feat(web): keep deployed WebUI out of search-engine indexes

Searching "TsmusicBot" surfaced many deployed instances' WebUI URLs,
letting strangers walk into other people's control pages (issue #128).
Add defence-in-depth so crawlers stop indexing public deployments:

- send `X-Robots-Tag: noindex, nofollow` on every Express response
- serve `/robots.txt` with `User-agent: * / Disallow: /`
- add `<meta name="robots" content="noindex, nofollow">` to index.html,
  which also covers the /bot/<id> dedicated-link pages (same SPA shell)

These layers only prevent indexing; real protection stays with WebUI
auth and the reverse proxy. Document this in the README security section
and warn users not to post their WebUI link on public pages.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
saopig1andClaude Fable 5 committed 2026-07-16 23:31:31 +08:00
1 parent 051171b019
commit ea0f7c17b5
4 files changed
+110 -3

No files matched your search

+86
View File
@@ -0,0 +1,86 @@
import { describe, it, expect } from "vitest";
import express from "express";
import request from "supertest";
import fs from "node:fs";
import path from "node:path";
import { fileURLToPath } from "node:url";
/**
* Search-engine hardening for issue #128: searching "TsmusicBot" surfaced a
* large number of deployed instances' WebUI URLs, letting strangers walk into
* other people's control pages. The fix is defence in depth — none of these
* layers is authentication (that's handled elsewhere), they just keep the
* public URL out of crawler indexes:
*
* 1. `X-Robots-Tag: noindex, nofollow` on EVERY response;
* 2. `GET /robots.txt` → `User-agent: * / Disallow: /`;
* 3. `<meta name="robots" content="noindex, nofollow">` in web/index.html.
*
* The header middleware and the /robots.txt route both live at the top of
* `createWebServer` in `server.ts`; this test asserts the exact behaviour we
* expect from them in isolation (the wiring inside server.ts is verified by
* code review / git diff, matching security-headers.test.ts).
*/
describe("search-engine hardening (issue #128 noindex)", () => {
function buildApp() {
const app = express();
// Mirrors the security-headers middleware in server.ts.
app.use((_req, res, next) => {
res.setHeader("X-Frame-Options", "DENY");
res.setHeader("Content-Security-Policy", "frame-ancestors 'none'");
res.setHeader("X-Robots-Tag", "noindex, nofollow");
next();
});
// Mirrors the public /robots.txt route in server.ts.
app.get("/robots.txt", (_req, res) => {
res.type("text/plain").send("User-agent: *\nDisallow: /\n");
});
app.get("/", (_req, res) => res.json({ ok: true }));
app.post("/api/session/login", (_req, res) => res.json({ ok: true }));
return app;
}
it("sets X-Robots-Tag: noindex, nofollow on GET responses", async () => {
const res = await request(buildApp()).get("/");
expect(res.status).toBe(200);
expect(res.headers["x-robots-tag"]).toBe("noindex, nofollow");
});
it("sets X-Robots-Tag on POST (API) responses too", async () => {
const res = await request(buildApp()).post("/api/session/login");
expect(res.headers["x-robots-tag"]).toBe("noindex, nofollow");
});
it("serves /robots.txt disallowing all crawlers", async () => {
const res = await request(buildApp()).get("/robots.txt");
expect(res.status).toBe(200);
expect(res.headers["content-type"]).toMatch(/text\/plain/);
expect(res.text).toContain("User-agent: *");
expect(res.text).toContain("Disallow: /");
});
it("still tags the /robots.txt response itself as noindex", async () => {
const res = await request(buildApp()).get("/robots.txt");
expect(res.headers["x-robots-tag"]).toBe("noindex, nofollow");
});
});
describe("frontend robots meta tag (issue #128 noindex)", () => {
const indexHtmlPath = path.resolve(
path.dirname(fileURLToPath(import.meta.url)),
"../../web/index.html"
);
const html = fs.readFileSync(indexHtmlPath, "utf-8");
const robotsMeta = html.match(
/<meta\s+name=["']robots["']\s+content=["']([^"']+)["']\s*\/?>/i
);
it("declares a robots meta tag", () => {
expect(robotsMeta).not.toBeNull();
});
it("marks the SPA shell noindex, nofollow (covers /bot/<id> dedicated links)", () => {
expect(robotsMeta?.[1]).toBe("noindex, nofollow");
});
});
+17 -3
View File
@@ -77,12 +77,19 @@ export function createWebServer(options: WebServerOptions): WebServer {
app.set("trust proxy", true);
}
// Security headers: prevent the WebUI from being embedded in a third-party
// iframe (clickjacking defence). CSP frame-ancestors is the modern equivalent
// of X-Frame-Options; both are set for compatibility across browsers.
// Security headers:
// • X-Frame-Options / CSP frame-ancestors — prevent the WebUI from being
// embedded in a third-party iframe (clickjacking defence). CSP
// frame-ancestors is the modern equivalent of X-Frame-Options; both are
// set for compatibility across browsers.
// • X-Robots-Tag — keep deployed instances out of search-engine indexes
// (issue #128: searching "TsmusicBot" surfaced strangers' WebUI URLs).
// Set on EVERY response so JSON/API responses and the SPA shell are all
// covered; complements /robots.txt and the <meta name="robots"> tag.
app.use((_req, res, next) => {
res.setHeader("X-Frame-Options", "DENY");
res.setHeader("Content-Security-Policy", "frame-ancestors 'none'");
res.setHeader("X-Robots-Tag", "noindex, nofollow");
next();
});
@@ -95,6 +102,13 @@ export function createWebServer(options: WebServerOptions): WebServer {
const permissions = createPermissionStore(options.database.db);
// ─── Public routes (no auth, no CSRF) ───────────────────────────────────
// Disallow every crawler (issue #128). Declared before the static SPA
// fallback so this wins over index.html for /robots.txt. Belt-and-braces
// with the X-Robots-Tag header above and the <meta name="robots"> tag.
app.get("/robots.txt", (_req, res) => {
res.type("text/plain").send("User-agent: *\nDisallow: /\n");
});
app.get("/api/health", (_req, res) => {
res.json({ status: "ok", version: "0.1.0" });
});