Files
gacha-event-tracker/test/robots.test.ts
T
Lucas WintherandClaude Opus 5 b085087b05 feat: refresh sources on a schedule
The feed was generated from checked-in fixtures and only moved when
somebody captured a page by hand. This fetches.

bun run refresh caches each page raw under snapshots/ and rebuilds the
feed from what it cached; build-feed prefers a snapshot and falls back to
the fixture, so a clean checkout and the container build stay offline and
reproducible. The workflow runs it twice a day and commits only when a
page's bytes actually changed — a 304, an identical body or a rejected
parse all leave the tree clean — then dispatches ci.yml, which already
knows how to test, build and deploy.

Scraping conduct is enforced in code rather than left to good
intentions: one request per source per cycle, a six-hour floor checked
per source, conditional requests, a User-Agent with a contact URL, and
robots.txt honoured — failing closed, because a permission we could not
read is not a permission we have. No retries; a retry is a second
request.

A body that yields zero events is rejected and the previous snapshot
kept, so a redesigned wiki shows up as a stale timestamp rather than an
emptied calendar. One source down is a warning; all of them down fails
the run, so a cycle that learned nothing is never committed.

Tested entirely offline against an injected fetch and clock — no request
has ever been made to a live wiki from this code.

Co-Authored-By: Claude Opus 5 (1M context) <[email protected]>
2026-08-15 21:15:25 +02:00

261 lines
8.2 KiB
TypeScript

import { describe, expect, test } from "bun:test";
import {
agentToken,
crawlDelayMs,
groupFor,
isAllowed,
parseRobots,
patternMatches,
requestTarget,
RobotsCache,
} from "../src/ingest/robots.ts";
const UA = "gacha-event-tracker/1.0 (+https://example.test/contact)";
describe("parseRobots", () => {
test("groups consecutive user-agent lines together", () => {
const robots = parseRobots(`
User-agent: alpha
User-agent: beta
Disallow: /private
User-agent: *
Disallow: /
`);
expect(robots.groups).toHaveLength(2);
expect(robots.groups[0]?.agents).toEqual(["alpha", "beta"]);
expect(robots.groups[1]?.agents).toEqual(["*"]);
});
test("ignores comments, blank lines and unknown directives", () => {
const robots = parseRobots(
"# a comment\r\nUser-agent: * # trailing\r\nHost: example.test\r\nDisallow: /x\r\nSitemap: https://example.test/sitemap.xml\r\n",
);
expect(robots.groups[0]?.rules).toEqual([{ allow: false, pattern: "/x" }]);
expect(robots.sitemaps).toEqual(["https://example.test/sitemap.xml"]);
});
test("an empty Disallow restricts nothing", () => {
const robots = parseRobots("User-agent: *\nDisallow:\n");
expect(robots.groups[0]?.rules).toEqual([]);
expect(isAllowed(robots, UA, "/anything")).toBe(true);
});
test("reads crawl-delay", () => {
const robots = parseRobots("User-agent: *\nCrawl-delay: 10\nDisallow: /x\n");
expect(crawlDelayMs(robots, UA)).toBe(10_000);
});
});
describe("group selection", () => {
test("takes the product token out of a full User-Agent header", () => {
expect(agentToken(UA)).toBe("gacha-event-tracker");
});
test("a named group beats the wildcard group", () => {
const robots = parseRobots(`
User-agent: *
Disallow: /
User-agent: gacha-event-tracker
Disallow: /admin
`);
expect(isAllowed(robots, UA, "/games/Genshin-Impact/archives/301601")).toBe(
true,
);
expect(isAllowed(robots, UA, "/admin/panel")).toBe(false);
});
test("falls back to the wildcard group when nothing names us", () => {
const robots = parseRobots("User-agent: *\nDisallow: /games/\n");
expect(isAllowed(robots, UA, "/games/x")).toBe(false);
});
test("no applicable group at all means allowed", () => {
const robots = parseRobots("User-agent: gptbot\nDisallow: /\n");
expect(groupFor(robots, UA)).toBeNull();
expect(isAllowed(robots, UA, "/games/x")).toBe(true);
});
test("the longest matching agent name wins", () => {
const robots = parseRobots(`
User-agent: googlebot
Disallow: /
User-agent: googlebot-news
Allow: /
`);
expect(isAllowed(robots, "Googlebot-News/1.0", "/anything")).toBe(true);
expect(isAllowed(robots, "Googlebot/2.1", "/anything")).toBe(false);
});
test("merges rules from several groups naming the same agent", () => {
const robots = parseRobots(`
User-agent: gacha-event-tracker
Disallow: /a
User-agent: gacha-event-tracker
Disallow: /b
`);
expect(isAllowed(robots, UA, "/a")).toBe(false);
expect(isAllowed(robots, UA, "/b")).toBe(false);
expect(isAllowed(robots, UA, "/c")).toBe(true);
});
});
describe("path matching", () => {
test("prefix match, wildcards and the end anchor", () => {
expect(patternMatches("/games/", "/games/Genshin")).toBe(true);
expect(patternMatches("/*.json", "/data/events.json")).toBe(true);
expect(patternMatches("/x$", "/x")).toBe(true);
expect(patternMatches("/x$", "/x/y")).toBe(false);
expect(patternMatches("", "/x")).toBe(false);
});
test("longest matching rule wins", () => {
const robots = parseRobots(`
User-agent: *
Disallow: /wiki/
Allow: /wiki/Event
`);
expect(isAllowed(robots, UA, "/wiki/Special:Random")).toBe(false);
expect(isAllowed(robots, UA, "/wiki/Event")).toBe(true);
});
test("allow wins a tie of equal length", () => {
const robots = parseRobots("User-agent: *\nDisallow: /page\nAllow: /page\n");
expect(isAllowed(robots, UA, "/page")).toBe(true);
});
test("the query string is part of the matched target", () => {
const robots = parseRobots("User-agent: *\nDisallow: /*?action=edit\n");
expect(requestTarget("https://x.test/wiki/Event?action=edit")).toBe(
"/wiki/Event?action=edit",
);
expect(isAllowed(robots, UA, "/wiki/Event?action=edit")).toBe(false);
expect(isAllowed(robots, UA, "/wiki/Event")).toBe(true);
});
test("a path is matched with a leading slash even if given without one", () => {
const robots = parseRobots("User-agent: *\nDisallow: /x\n");
expect(isAllowed(robots, UA, "x")).toBe(false);
});
});
describe("the sources we actually fetch", () => {
// Game8 opts out of AI-training crawlers by name and leaves everyone else
// alone (CLAUDE.md § Scraping conduct). If that ever changes, this is where
// it should be noticed.
const game8 = parseRobots(`
User-agent: GPTBot
Disallow: /
User-agent: Google-Extended
Disallow: /
User-agent: *
Disallow: /admin/
Disallow: /*?utm_source=
`);
test("our agent may fetch a Game8 article page", () => {
expect(
isAllowed(game8, UA, "/games/Genshin-Impact/archives/301601"),
).toBe(true);
});
test("the AI-training opt-outs still bind those crawlers", () => {
expect(isAllowed(game8, "GPTBot/1.2", "/games/x")).toBe(false);
expect(isAllowed(game8, "Google-Extended", "/games/x")).toBe(false);
});
test("the wildcard rules that do exist are obeyed", () => {
expect(isAllowed(game8, UA, "/admin/")).toBe(false);
expect(isAllowed(game8, UA, "/games/x?utm_source=y")).toBe(false);
});
});
describe("RobotsCache", () => {
function cacheWith(
responder: (url: string) => Response | Promise<Response>,
calls: string[] = [],
) {
return {
calls,
cache: new RobotsCache({
userAgent: UA,
fetchImpl: async (url) => {
calls.push(url);
return responder(url);
},
}),
};
}
test("fetches robots.txt once per host and reuses it", async () => {
const { cache, calls } = cacheWith(
() => new Response("User-agent: *\nDisallow: /admin\n", { status: 200 }),
);
expect((await cache.allows("https://game8.co/games/a")).allowed).toBe(true);
expect((await cache.allows("https://game8.co/games/b")).allowed).toBe(true);
expect((await cache.allows("https://game8.co/admin")).allowed).toBe(false);
expect(calls).toEqual(["https://game8.co/robots.txt"]);
expect(cache.fetches).toBe(1);
});
test("fetches once per distinct host", async () => {
const { cache, calls } = cacheWith(() => new Response("", { status: 200 }));
await cache.allows("https://game8.co/a");
await cache.allows("https://endfield.wiki.gg/wiki/Event");
expect(calls).toEqual([
"https://game8.co/robots.txt",
"https://endfield.wiki.gg/robots.txt",
]);
});
test("a missing robots.txt means no restrictions", async () => {
const { cache } = cacheWith(() => new Response("nope", { status: 404 }));
const decision = await cache.allows("https://x.test/wiki/Event");
expect(decision.allowed).toBe(true);
expect(decision.reason).toBe("no robots.txt");
});
test("fails closed on a server error", async () => {
const { cache } = cacheWith(() => new Response("", { status: 503 }));
const decision = await cache.allows("https://x.test/wiki/Event");
expect(decision.allowed).toBe(false);
expect(decision.reason).toContain("503");
});
test("fails closed when robots.txt is unreachable", async () => {
const { cache } = cacheWith(() => {
throw new Error("ECONNREFUSED");
});
const decision = await cache.allows("https://x.test/wiki/Event");
expect(decision.allowed).toBe(false);
expect(decision.reason).toContain("unreachable");
});
test("expires an entry after its TTL", async () => {
const calls: string[] = [];
let clock = 0;
const cache = new RobotsCache({
userAgent: UA,
fetchImpl: async (url) => {
calls.push(url);
return new Response("User-agent: *\nDisallow:\n", { status: 200 });
},
ttlMs: 1000,
now: () => clock,
});
await cache.allows("https://x.test/a");
clock = 999;
await cache.allows("https://x.test/a");
expect(calls).toHaveLength(1);
clock = 1001;
await cache.allows("https://x.test/a");
expect(calls).toHaveLength(2);
});
});