The feed was generated from checked-in fixtures and only moved when somebody captured a page by hand. This fetches. bun run refresh caches each page raw under snapshots/ and rebuilds the feed from what it cached; build-feed prefers a snapshot and falls back to the fixture, so a clean checkout and the container build stay offline and reproducible. The workflow runs it twice a day and commits only when a page's bytes actually changed — a 304, an identical body or a rejected parse all leave the tree clean — then dispatches ci.yml, which already knows how to test, build and deploy. Scraping conduct is enforced in code rather than left to good intentions: one request per source per cycle, a six-hour floor checked per source, conditional requests, a User-Agent with a contact URL, and robots.txt honoured — failing closed, because a permission we could not read is not a permission we have. No retries; a retry is a second request. A body that yields zero events is rejected and the previous snapshot kept, so a redesigned wiki shows up as a stale timestamp rather than an emptied calendar. One source down is a warning; all of them down fails the run, so a cycle that learned nothing is never committed. Tested entirely offline against an injected fetch and clock — no request has ever been made to a live wiki from this code. Co-Authored-By: Claude Opus 5 (1M context) <[email protected]>
261 lines
8.2 KiB
TypeScript
261 lines
8.2 KiB
TypeScript
import { describe, expect, test } from "bun:test";
|
|
import {
|
|
agentToken,
|
|
crawlDelayMs,
|
|
groupFor,
|
|
isAllowed,
|
|
parseRobots,
|
|
patternMatches,
|
|
requestTarget,
|
|
RobotsCache,
|
|
} from "../src/ingest/robots.ts";
|
|
|
|
const UA = "gacha-event-tracker/1.0 (+https://example.test/contact)";
|
|
|
|
describe("parseRobots", () => {
|
|
test("groups consecutive user-agent lines together", () => {
|
|
const robots = parseRobots(`
|
|
User-agent: alpha
|
|
User-agent: beta
|
|
Disallow: /private
|
|
|
|
User-agent: *
|
|
Disallow: /
|
|
`);
|
|
expect(robots.groups).toHaveLength(2);
|
|
expect(robots.groups[0]?.agents).toEqual(["alpha", "beta"]);
|
|
expect(robots.groups[1]?.agents).toEqual(["*"]);
|
|
});
|
|
|
|
test("ignores comments, blank lines and unknown directives", () => {
|
|
const robots = parseRobots(
|
|
"# a comment\r\nUser-agent: * # trailing\r\nHost: example.test\r\nDisallow: /x\r\nSitemap: https://example.test/sitemap.xml\r\n",
|
|
);
|
|
expect(robots.groups[0]?.rules).toEqual([{ allow: false, pattern: "/x" }]);
|
|
expect(robots.sitemaps).toEqual(["https://example.test/sitemap.xml"]);
|
|
});
|
|
|
|
test("an empty Disallow restricts nothing", () => {
|
|
const robots = parseRobots("User-agent: *\nDisallow:\n");
|
|
expect(robots.groups[0]?.rules).toEqual([]);
|
|
expect(isAllowed(robots, UA, "/anything")).toBe(true);
|
|
});
|
|
|
|
test("reads crawl-delay", () => {
|
|
const robots = parseRobots("User-agent: *\nCrawl-delay: 10\nDisallow: /x\n");
|
|
expect(crawlDelayMs(robots, UA)).toBe(10_000);
|
|
});
|
|
});
|
|
|
|
describe("group selection", () => {
|
|
test("takes the product token out of a full User-Agent header", () => {
|
|
expect(agentToken(UA)).toBe("gacha-event-tracker");
|
|
});
|
|
|
|
test("a named group beats the wildcard group", () => {
|
|
const robots = parseRobots(`
|
|
User-agent: *
|
|
Disallow: /
|
|
|
|
User-agent: gacha-event-tracker
|
|
Disallow: /admin
|
|
`);
|
|
expect(isAllowed(robots, UA, "/games/Genshin-Impact/archives/301601")).toBe(
|
|
true,
|
|
);
|
|
expect(isAllowed(robots, UA, "/admin/panel")).toBe(false);
|
|
});
|
|
|
|
test("falls back to the wildcard group when nothing names us", () => {
|
|
const robots = parseRobots("User-agent: *\nDisallow: /games/\n");
|
|
expect(isAllowed(robots, UA, "/games/x")).toBe(false);
|
|
});
|
|
|
|
test("no applicable group at all means allowed", () => {
|
|
const robots = parseRobots("User-agent: gptbot\nDisallow: /\n");
|
|
expect(groupFor(robots, UA)).toBeNull();
|
|
expect(isAllowed(robots, UA, "/games/x")).toBe(true);
|
|
});
|
|
|
|
test("the longest matching agent name wins", () => {
|
|
const robots = parseRobots(`
|
|
User-agent: googlebot
|
|
Disallow: /
|
|
|
|
User-agent: googlebot-news
|
|
Allow: /
|
|
`);
|
|
expect(isAllowed(robots, "Googlebot-News/1.0", "/anything")).toBe(true);
|
|
expect(isAllowed(robots, "Googlebot/2.1", "/anything")).toBe(false);
|
|
});
|
|
|
|
test("merges rules from several groups naming the same agent", () => {
|
|
const robots = parseRobots(`
|
|
User-agent: gacha-event-tracker
|
|
Disallow: /a
|
|
|
|
User-agent: gacha-event-tracker
|
|
Disallow: /b
|
|
`);
|
|
expect(isAllowed(robots, UA, "/a")).toBe(false);
|
|
expect(isAllowed(robots, UA, "/b")).toBe(false);
|
|
expect(isAllowed(robots, UA, "/c")).toBe(true);
|
|
});
|
|
});
|
|
|
|
describe("path matching", () => {
|
|
test("prefix match, wildcards and the end anchor", () => {
|
|
expect(patternMatches("/games/", "/games/Genshin")).toBe(true);
|
|
expect(patternMatches("/*.json", "/data/events.json")).toBe(true);
|
|
expect(patternMatches("/x$", "/x")).toBe(true);
|
|
expect(patternMatches("/x$", "/x/y")).toBe(false);
|
|
expect(patternMatches("", "/x")).toBe(false);
|
|
});
|
|
|
|
test("longest matching rule wins", () => {
|
|
const robots = parseRobots(`
|
|
User-agent: *
|
|
Disallow: /wiki/
|
|
Allow: /wiki/Event
|
|
`);
|
|
expect(isAllowed(robots, UA, "/wiki/Special:Random")).toBe(false);
|
|
expect(isAllowed(robots, UA, "/wiki/Event")).toBe(true);
|
|
});
|
|
|
|
test("allow wins a tie of equal length", () => {
|
|
const robots = parseRobots("User-agent: *\nDisallow: /page\nAllow: /page\n");
|
|
expect(isAllowed(robots, UA, "/page")).toBe(true);
|
|
});
|
|
|
|
test("the query string is part of the matched target", () => {
|
|
const robots = parseRobots("User-agent: *\nDisallow: /*?action=edit\n");
|
|
expect(requestTarget("https://x.test/wiki/Event?action=edit")).toBe(
|
|
"/wiki/Event?action=edit",
|
|
);
|
|
expect(isAllowed(robots, UA, "/wiki/Event?action=edit")).toBe(false);
|
|
expect(isAllowed(robots, UA, "/wiki/Event")).toBe(true);
|
|
});
|
|
|
|
test("a path is matched with a leading slash even if given without one", () => {
|
|
const robots = parseRobots("User-agent: *\nDisallow: /x\n");
|
|
expect(isAllowed(robots, UA, "x")).toBe(false);
|
|
});
|
|
});
|
|
|
|
describe("the sources we actually fetch", () => {
|
|
// Game8 opts out of AI-training crawlers by name and leaves everyone else
|
|
// alone (CLAUDE.md § Scraping conduct). If that ever changes, this is where
|
|
// it should be noticed.
|
|
const game8 = parseRobots(`
|
|
User-agent: GPTBot
|
|
Disallow: /
|
|
|
|
User-agent: Google-Extended
|
|
Disallow: /
|
|
|
|
User-agent: *
|
|
Disallow: /admin/
|
|
Disallow: /*?utm_source=
|
|
`);
|
|
|
|
test("our agent may fetch a Game8 article page", () => {
|
|
expect(
|
|
isAllowed(game8, UA, "/games/Genshin-Impact/archives/301601"),
|
|
).toBe(true);
|
|
});
|
|
|
|
test("the AI-training opt-outs still bind those crawlers", () => {
|
|
expect(isAllowed(game8, "GPTBot/1.2", "/games/x")).toBe(false);
|
|
expect(isAllowed(game8, "Google-Extended", "/games/x")).toBe(false);
|
|
});
|
|
|
|
test("the wildcard rules that do exist are obeyed", () => {
|
|
expect(isAllowed(game8, UA, "/admin/")).toBe(false);
|
|
expect(isAllowed(game8, UA, "/games/x?utm_source=y")).toBe(false);
|
|
});
|
|
});
|
|
|
|
describe("RobotsCache", () => {
|
|
function cacheWith(
|
|
responder: (url: string) => Response | Promise<Response>,
|
|
calls: string[] = [],
|
|
) {
|
|
return {
|
|
calls,
|
|
cache: new RobotsCache({
|
|
userAgent: UA,
|
|
fetchImpl: async (url) => {
|
|
calls.push(url);
|
|
return responder(url);
|
|
},
|
|
}),
|
|
};
|
|
}
|
|
|
|
test("fetches robots.txt once per host and reuses it", async () => {
|
|
const { cache, calls } = cacheWith(
|
|
() => new Response("User-agent: *\nDisallow: /admin\n", { status: 200 }),
|
|
);
|
|
expect((await cache.allows("https://game8.co/games/a")).allowed).toBe(true);
|
|
expect((await cache.allows("https://game8.co/games/b")).allowed).toBe(true);
|
|
expect((await cache.allows("https://game8.co/admin")).allowed).toBe(false);
|
|
expect(calls).toEqual(["https://game8.co/robots.txt"]);
|
|
expect(cache.fetches).toBe(1);
|
|
});
|
|
|
|
test("fetches once per distinct host", async () => {
|
|
const { cache, calls } = cacheWith(() => new Response("", { status: 200 }));
|
|
await cache.allows("https://game8.co/a");
|
|
await cache.allows("https://endfield.wiki.gg/wiki/Event");
|
|
expect(calls).toEqual([
|
|
"https://game8.co/robots.txt",
|
|
"https://endfield.wiki.gg/robots.txt",
|
|
]);
|
|
});
|
|
|
|
test("a missing robots.txt means no restrictions", async () => {
|
|
const { cache } = cacheWith(() => new Response("nope", { status: 404 }));
|
|
const decision = await cache.allows("https://x.test/wiki/Event");
|
|
expect(decision.allowed).toBe(true);
|
|
expect(decision.reason).toBe("no robots.txt");
|
|
});
|
|
|
|
test("fails closed on a server error", async () => {
|
|
const { cache } = cacheWith(() => new Response("", { status: 503 }));
|
|
const decision = await cache.allows("https://x.test/wiki/Event");
|
|
expect(decision.allowed).toBe(false);
|
|
expect(decision.reason).toContain("503");
|
|
});
|
|
|
|
test("fails closed when robots.txt is unreachable", async () => {
|
|
const { cache } = cacheWith(() => {
|
|
throw new Error("ECONNREFUSED");
|
|
});
|
|
const decision = await cache.allows("https://x.test/wiki/Event");
|
|
expect(decision.allowed).toBe(false);
|
|
expect(decision.reason).toContain("unreachable");
|
|
});
|
|
|
|
test("expires an entry after its TTL", async () => {
|
|
const calls: string[] = [];
|
|
let clock = 0;
|
|
const cache = new RobotsCache({
|
|
userAgent: UA,
|
|
fetchImpl: async (url) => {
|
|
calls.push(url);
|
|
return new Response("User-agent: *\nDisallow:\n", { status: 200 });
|
|
},
|
|
ttlMs: 1000,
|
|
now: () => clock,
|
|
});
|
|
|
|
await cache.allows("https://x.test/a");
|
|
clock = 999;
|
|
await cache.allows("https://x.test/a");
|
|
expect(calls).toHaveLength(1);
|
|
clock = 1001;
|
|
await cache.allows("https://x.test/a");
|
|
expect(calls).toHaveLength(2);
|
|
});
|
|
});
|