diff --git a/.github/workflows/refresh.yml b/.github/workflows/refresh.yml new file mode 100644 index 0000000..7b10316 --- /dev/null +++ b/.github/workflows/refresh.yml @@ -0,0 +1,128 @@ +name: Refresh sources + +# Fetch each source at most twice a day and commit the raw snapshots when — and +# only when — the bytes actually changed. Everything downstream (parse, merge, +# feed, build, deploy) is CI's existing job; this workflow does not duplicate +# any of it, it just hands CI fresher input. +# +# Twelve hours apart is deliberately well clear of the six-hour-per-source floor +# in CLAUDE.md § Scraping conduct, and the runner enforces that floor itself, so +# a manual dispatch on top of a scheduled run cannot double up on a wiki. +on: + schedule: + - cron: "27 5,17 * * *" + workflow_dispatch: + inputs: + dry_run: + description: "Plan only — no requests, no writes" + type: boolean + default: false + only: + description: "Refresh a single source id (blank = all)" + type: string + default: "" + +# Never two refreshes at once: they would both fetch, and the second would race +# the first's commit. Queue instead of cancelling — a half-finished refresh that +# has already written snapshots should be allowed to finish and push. +concurrency: + group: refresh + cancel-in-progress: false + +permissions: + contents: read + +jobs: + refresh: + name: Fetch sources and rebuild the feed + runs-on: ubuntu-latest + # A fork must not point this at the wikis on a schedule. Same shape as the + # DEPLOY_PAGES gate in ci.yml: off by default for anyone but this repo, + # while a fork owner can still dispatch it by hand and take responsibility. + if: >- + github.event_name == 'workflow_dispatch' || + github.repository == 'StereotypicalCat/gacha-event-tracker' + permissions: + # Commit the refreshed snapshots. + contents: write + # Dispatch ci.yml afterwards: a push made with GITHUB_TOKEN deliberately + # does not trigger other workflows, so without this the fresh data would + # sit in the repo undeployed until someone pushed by hand. + actions: write + steps: + - uses: actions/checkout@v4 + + - uses: oven-sh/setup-bun@v2 + with: + bun-version: "1.3" + + - run: bun install --frozen-lockfile + + # When each source was last checked. Gitignored on purpose (committing it + # would mean a commit every cycle saying nothing changed), so it rides in + # the actions cache instead. A cache miss only means the runner has no + # record of the last check — the twelve-hour schedule still keeps us well + # inside the etiquette floor. + - name: Restore refresh bookkeeping + uses: actions/cache/restore@v4 + with: + path: snapshots/*.state.json + key: refresh-state-${{ github.run_id }} + restore-keys: refresh-state- + + - name: Refresh + env: + # Identify the crawler with a contact URL, per CLAUDE.md. + REFRESH_CONTACT_URL: ${{ github.server_url }}/${{ github.repository }} + # Passed through the environment rather than interpolated into the + # run script, so a dispatch input cannot become shell. + ONLY: ${{ inputs.only }} + DRY_RUN: ${{ inputs.dry_run }} + run: | + args=() + if [ "$DRY_RUN" = "true" ]; then + args+=(--dry-run) + fi + if [ -n "$ONLY" ]; then + args+=(--only "$ONLY") + fi + bun run refresh "${args[@]}" + + - name: Save refresh bookkeeping + if: always() + uses: actions/cache/save@v4 + with: + path: snapshots/*.state.json + key: refresh-state-${{ github.run_id }} + + # git is the authority on "did anything change" — a 304, an unchanged + # body, or a rejected parse all leave the working tree clean. + - name: Detect changes + id: diff + run: | + if [ -n "$(git status --porcelain -- snapshots)" ]; then + git status --porcelain -- snapshots + echo "changed=true" >> "$GITHUB_OUTPUT" + else + echo "no source changed" + echo "changed=false" >> "$GITHUB_OUTPUT" + fi + + - name: Commit refreshed snapshots + if: steps.diff.outputs.changed == 'true' && inputs.dry_run != true + run: | + git config user.name "github-actions[bot]" + git config user.email "41898282+github-actions[bot]@users.noreply.github.com" + git add -- snapshots + git commit -m "chore(data): refresh source snapshots" \ + -m "Automated fetch from ${{ github.workflow }} run ${{ github.run_id }}." + git push + + # ci.yml owns typecheck, tests, the feed sanity check, the image and the + # Pages deploy. Dispatching it is how the refreshed data reaches the site + # without any of that logic being copied here. + - name: Publish the refreshed feed + if: steps.diff.outputs.changed == 'true' && inputs.dry_run != true + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: gh workflow run ci.yml --ref "${{ github.ref_name }}" diff --git a/.gitignore b/.gitignore index 4871830..ad0f893 100644 --- a/.gitignore +++ b/.gitignore @@ -120,3 +120,9 @@ data/ # build output public/ + +# refresh bookkeeping (when we last checked a source). +# The snapshots themselves ARE tracked — they are what the feed is built from — +# but this file changes on every cycle, and committing it would mean a commit +# per run saying nothing changed. +snapshots/*.state.json diff --git a/Dockerfile b/Dockerfile index 98476ae..591c210 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,9 +1,10 @@ # Build the static site, then serve it from a distroless-ish runtime. # # Two stages so the image ships the built assets and nothing else: no source, -# no fixtures, no toolchain. The build is fully offline — it parses checked-in -# fixtures rather than fetching anything — so the image is reproducible and -# needs no network at build time. +# no fixtures, no toolchain. The build is fully offline — it parses the +# committed snapshots, falling back to checked-in fixtures, rather than +# fetching anything — so the image is reproducible and needs no network at +# build time. FROM oven/bun:1.3-alpine AS build WORKDIR /app @@ -16,6 +17,12 @@ COPY tsconfig.json index.html serve.ts ./ COPY src ./src COPY scripts ./scripts COPY fixtures ./fixtures +# Whatever the last refresh committed. The directory always exists (it carries +# a README), so this cannot break a build made before the first refresh — it +# just leaves build:feed on the fixture fallback, which is what the image did +# before. Without it the container would serve fixture-era data while the site +# served fresh, with nothing to say why. +COPY snapshots ./snapshots COPY test ./test # Fail the image on a type error or a failing test rather than shipping it. diff --git a/package.json b/package.json index 14e25a7..be1b121 100644 --- a/package.json +++ b/package.json @@ -7,6 +7,7 @@ "typecheck": "tsc --noEmit", "parse": "bun run scripts/parse-fixture.ts", "build:feed": "bun run scripts/build-feed.ts", + "refresh": "bun run scripts/refresh-sources.ts", "build:css": "bunx @tailwindcss/cli -i src/client/styles.css -o public/styles.css --minify", "build:js": "bun build src/client/main.tsx --outfile public/main.js --minify", "build": "bun run build:feed && bun run build:css && bun run build:js && bun run build:static", diff --git a/scripts/build-feed.ts b/scripts/build-feed.ts index 8f750d2..59e73c0 100644 --- a/scripts/build-feed.ts +++ b/scripts/build-feed.ts @@ -1,18 +1,22 @@ /** - * Build the static event feed from checked-in fixtures. + * Build the static event feed from cached snapshots, falling back to fixtures. * - * Offline: this reads fixtures, never the network. It exists so the client can - * be developed and demoed against real parsed data before the server and - * database land, and it emits exactly the shape `GET /api/events.json` will. + * Offline: this reads files on disk, never the network. Fetching is + * `scripts/refresh-sources.ts`'s job; this stage only parses what that left in + * the snapshot cache. On a clean checkout — and in the container build — no + * snapshot exists and the checked-in fixture is used instead, so the build + * stays reproducible and a wiki being down never breaks it. * * bun run build:feed */ import { ADAPTERS } from "../src/ingest/adapters/index.ts"; import { mergeEvents } from "../src/ingest/merge.ts"; +import { SnapshotStore, freshnessAt } from "../src/ingest/snapshots.ts"; import { EventFeed, SCHEMA_VERSION, type SourceHealth } from "../src/shared/feed.ts"; import type { GachaEvent, GameId } from "../src/shared/schema.ts"; const OUT = "public/data/events.v1.json"; +const snapshots = new SnapshotStore(process.env["SNAPSHOT_DIR"] ?? "snapshots"); /** * Newest fixture for one *source*, not one game. @@ -32,12 +36,32 @@ async function latestFixture(adapterId: string, game: GameId) { return { file, html: await Bun.file(file).text() }; } +/** + * The document to parse for one source: the live snapshot when the refresh + * runner has cached one, otherwise the newest checked-in fixture. + * + * `at` is what the UI's staleness badge reads, so it must never claim to be + * fresher than the bytes actually are — a fixture reports its capture date. + */ +async function documentFor(adapterId: string, game: GameId) { + const cached = await snapshots.read(adapterId); + if (cached !== null) { + return { + file: snapshots.bodyPath(adapterId), + html: cached.html, + at: freshnessAt(cached), + }; + } + const { file, html } = await latestFixture(adapterId, game); + return { file, html, at: fixtureDate(file) }; +} + const now = new Date().toISOString(); const byGame = new Map(); const sources: SourceHealth[] = []; for (const adapter of ADAPTERS) { - const { file, html } = await latestFixture(adapter.id, adapter.game); + const { file, html, at } = await documentFor(adapter.id, adapter.game); const events = adapter.parse(html, { now, sourceUrl: adapter.url, @@ -53,10 +77,9 @@ for (const adapter of ADAPTERS) { sourceId: adapter.id, game: adapter.game, url: adapter.url, - // Fixture capture date stands in for a real fetch timestamp until the - // scheduler exists. The UI's staleness badge reads this, so it must not - // claim to be fresher than the data actually is. - lastSuccessAt: fixtureDate(file), + // When the bytes were last confirmed live; a fixture's capture date when + // this source has never been refreshed. + lastSuccessAt: at, eventCount: events.length, }); diff --git a/scripts/refresh-sources.ts b/scripts/refresh-sources.ts new file mode 100644 index 0000000..ca7201d --- /dev/null +++ b/scripts/refresh-sources.ts @@ -0,0 +1,492 @@ +/** + * Refresh every source, then regenerate the feed. + * + * bun run refresh # the real thing + * bun run refresh --dry-run # plan only, no requests, no writes + * bun run refresh --only genshin-game8-events + * + * This is the scheduled half of the pipeline (docs/INGESTION.md stages 1-2). + * The rules it enforces are etiquette obligations, not preferences: + * + * - robots.txt is read once per host per run and obeyed; unreadable means + * "do not fetch", never "assume yes". + * - at most ONE request per source per cycle, and never sooner than six hours + * after the last attempt. There is deliberately no retry: a retry is a + * second request, and the next cycle is minutes-cheap compared to being a + * bad guest. + * - conditional requests always, so an unchanged page costs the wiki a 304. + * - a descriptive User-Agent carrying a contact URL. + * + * Failure policy: one wiki being down is a warning. The previous snapshot stays + * in place and the feed keeps its events — a source outage must never blank the + * calendar. Hard failures (bad arguments, an unwritable cache, a feed that will + * not rebuild) exit non-zero so CI stops before committing anything. + */ +import { + ADAPTERS, + adapterById, +} from "../src/ingest/adapters/index.ts"; +import { SIX_HOURS_MS } from "../src/ingest/adapters/types.ts"; +import type { Adapter } from "../src/ingest/adapters/types.ts"; +import { RobotsCache, type FetchLike } from "../src/ingest/robots.ts"; +import { SnapshotStore } from "../src/ingest/snapshots.ts"; + +const DEFAULT_CONTACT = + "https://github.com/StereotypicalCat/gacha-event-tracker"; + +export const DEFAULT_USER_AGENT = `gacha-event-tracker/1.0 (+${process.env["REFRESH_CONTACT_URL"] ?? DEFAULT_CONTACT})`; + +/** How a single source's cycle ended. */ +export type RefreshResult = + | "fetched" // 200 with new bytes, parsed, stored + | "unchanged" // 304, or 200 whose bytes matched what we had + | "skipped_interval" // fetched too recently to ask again + | "skipped_robots" // robots.txt says no, or could not be read + | "rejected" // fetched, but the body parsed worse than what we hold + | "failed" // unreachable or an error status + | "planned"; // --dry-run + +export interface SourceOutcome { + sourceId: string; + result: RefreshResult; + note: string; + status: number | null; + eventCount: number | null; +} + +export interface RefreshSummary { + outcomes: SourceOutcome[]; + /** Sources whose stored bytes changed — the only reason to commit. */ + changed: number; + /** Sources we actually sent a request to. */ + attempted: number; + /** Sources that answered (200 or 304). */ + confirmed: number; + warnings: string[]; + /** Set when the run should exit non-zero. */ + hardFailure: string | null; +} + +export interface RobotsGate { + allows(url: string): Promise<{ allowed: boolean; reason: string }>; +} + +export interface RefreshOptions { + adapters: readonly Adapter[]; + store: SnapshotStore; + robots: RobotsGate; + fetchImpl: FetchLike; + userAgent: string; + /** Injected clock — the runner is testable, like the parsers it drives. */ + now: () => Date; + dryRun: boolean; + only: string | null; + timeoutMs: number; + log: (line: string) => void; + /** Called once when something changed. Null skips the rebuild (tests). */ + rebuildFeed: (() => Promise) | null; +} + +/** A drop this steep means the page changed shape, not that events ended. */ +const DROP_WARNING_RATIO = 0.5; + +export async function runRefresh( + options: RefreshOptions, +): Promise { + const summary: RefreshSummary = { + outcomes: [], + changed: 0, + attempted: 0, + confirmed: 0, + warnings: [], + hardFailure: null, + }; + + const selected = + options.only === null + ? [...options.adapters] + : options.adapters.filter((a) => a.id === options.only); + + if (selected.length === 0) { + summary.hardFailure = `unknown source '${options.only ?? ""}'`; + return summary; + } + + for (const adapter of selected) { + const outcome = await refreshOne(adapter, options); + summary.outcomes.push(outcome); + + if (outcome.result === "fetched") { + summary.changed += 1; + summary.attempted += 1; + summary.confirmed += 1; + } else if (outcome.result === "unchanged") { + summary.attempted += 1; + summary.confirmed += 1; + } else if (outcome.result === "failed" || outcome.result === "rejected") { + summary.attempted += 1; + summary.warnings.push(`${adapter.id}: ${outcome.note}`); + } else if (outcome.result === "skipped_robots") { + summary.warnings.push(`${adapter.id}: ${outcome.note}`); + } + + options.log( + ` ${adapter.id.padEnd(24)} ${outcome.result.padEnd(17)} ${outcome.note}`, + ); + } + + // Every source failing is not "a wiki is down", it is us: no network, a bad + // User-Agent, a proxy. That should stop the pipeline rather than look green. + if (summary.attempted > 0 && summary.confirmed === 0) { + summary.hardFailure = `all ${summary.attempted} attempted sources failed`; + return summary; + } + + // Likewise, being turned away everywhere is news. Left as a warning it would + // read as a quiet, successful, permanently empty refresh. + if (summary.outcomes.every((o) => o.result === "skipped_robots")) { + summary.hardFailure = `robots.txt blocked all ${summary.outcomes.length} sources`; + return summary; + } + + if (summary.changed > 0 && options.rebuildFeed !== null) { + try { + await options.rebuildFeed(); + } catch (error) { + // New snapshots are on disk but do not produce a feed. Exiting non-zero + // keeps CI from committing them. + summary.hardFailure = `feed rebuild failed: ${String(error)}`; + } + } + + return summary; +} + +async function refreshOne( + adapter: Adapter, + options: RefreshOptions, +): Promise { + const { store } = options; + const now = options.now(); + const nowIso = now.toISOString(); + const meta = await store.readMeta(adapter.id); + const state = await store.readState(adapter.id); + const headers = store.conditionalHeaders(meta); + + if (!store.isDue(state, now.getTime(), adapter.minIntervalMs)) { + const dueAt = new Date(store.dueAt(state, adapter.minIntervalMs)); + return { + sourceId: adapter.id, + result: "skipped_interval", + note: `checked ${state.lastCheckedAt ?? "?"}, next due ${dueAt.toISOString()}`, + status: null, + eventCount: meta?.eventCount ?? null, + }; + } + + if (options.dryRun) { + const conditional = Object.keys(headers); + return { + sourceId: adapter.id, + result: "planned", + note: `would GET ${adapter.url}${ + conditional.length > 0 ? ` with ${conditional.join(", ")}` : " (no validators cached)" + }`, + status: null, + eventCount: meta?.eventCount ?? null, + }; + } + + const decision = await options.robots.allows(adapter.url); + if (!decision.allowed) { + return { + sourceId: adapter.id, + result: "skipped_robots", + note: decision.reason, + status: null, + eventCount: meta?.eventCount ?? null, + }; + } + + let response: Response; + try { + response = await options.fetchImpl(adapter.url, { + headers: { + "User-Agent": options.userAgent, + Accept: "text/html,application/xhtml+xml", + ...headers, + }, + signal: AbortSignal.timeout(options.timeoutMs), + redirect: "follow", + }); + } catch (error) { + await store.recordCheck(adapter.id, { at: nowIso, status: null, ok: false }); + return { + sourceId: adapter.id, + result: "failed", + note: `unreachable: ${String(error)}`, + status: null, + eventCount: meta?.eventCount ?? null, + }; + } + + if (response.status === 304) { + await store.recordCheck(adapter.id, { at: nowIso, status: 304, ok: true }); + return { + sourceId: adapter.id, + result: "unchanged", + note: "304 not modified", + status: 304, + eventCount: meta?.eventCount ?? null, + }; + } + + if (!response.ok) { + await store.recordCheck(adapter.id, { + at: nowIso, + status: response.status, + ok: false, + }); + return { + sourceId: adapter.id, + result: "failed", + note: `HTTP ${response.status}`, + status: response.status, + eventCount: meta?.eventCount ?? null, + }; + } + + const html = await response.text(); + + // The parse gate. A body that no longer parses, or that yields nothing where + // it used to yield events, is a source that changed shape — publishing it + // would empty a game's calendar silently, which is the failure this pipeline + // exists to avoid. Keep what we hold and warn. + let events: number; + try { + events = adapter.parse(html, { + now: nowIso, + sourceUrl: adapter.url, + sourceId: adapter.id, + game: adapter.game, + }).length; + } catch (error) { + await store.recordCheck(adapter.id, { + at: nowIso, + status: response.status, + ok: false, + }); + return { + sourceId: adapter.id, + result: "rejected", + note: `kept previous snapshot; new body did not parse: ${String(error)}`, + status: response.status, + eventCount: meta?.eventCount ?? null, + }; + } + + // Zero events is never a useful snapshot: every source in the registry + // yields events by construction, so an empty parse means the page changed + // shape. Refusing it keeps the previous snapshot — or, on a first run, the + // checked-in fixture — as the thing the feed is built from. + const previousCount = meta?.eventCount ?? null; + if (events === 0) { + await store.recordCheck(adapter.id, { + at: nowIso, + status: response.status, + ok: false, + }); + return { + sourceId: adapter.id, + result: "rejected", + note: + previousCount === null + ? "did not store; body yielded 0 events" + : `kept previous snapshot; new body yielded 0 events (had ${previousCount})`, + status: response.status, + eventCount: previousCount, + }; + } + + const saved = await store.save(adapter.id, { + url: adapter.url, + html, + etag: response.headers.get("ETag"), + lastModified: response.headers.get("Last-Modified"), + at: nowIso, + eventCount: events, + }); + await store.recordCheck(adapter.id, { + at: nowIso, + status: response.status, + ok: true, + }); + + if (!saved.changed) { + return { + sourceId: adapter.id, + result: "unchanged", + note: `200 but identical bytes (${events} events)`, + status: response.status, + eventCount: events, + }; + } + + const dropped = + previousCount !== null && + previousCount > 0 && + events < previousCount * DROP_WARNING_RATIO; + + return { + sourceId: adapter.id, + result: "fetched", + note: dropped + ? `${events} events — down from ${previousCount}, check the page shape` + : `${events} events`, + status: response.status, + eventCount: events, + }; +} + +/** Regenerate public/data/events.v1.json from whatever is now cached. */ +export async function rebuildFeedViaScript(): Promise { + const proc = Bun.spawn(["bun", "run", "scripts/build-feed.ts"], { + stdout: "inherit", + stderr: "inherit", + }); + const code = await proc.exited; + if (code !== 0) throw new Error(`build-feed exited ${code}`); +} + +interface Args { + dryRun: boolean; + only: string | null; + root: string; + userAgent: string; + rebuild: boolean; + help: boolean; +} + +export function parseArgs(argv: readonly string[]): Args { + const args: Args = { + dryRun: false, + only: null, + root: process.env["SNAPSHOT_DIR"] ?? "snapshots", + userAgent: process.env["REFRESH_USER_AGENT"] ?? DEFAULT_USER_AGENT, + rebuild: true, + help: false, + }; + + for (let i = 0; i < argv.length; i += 1) { + const arg = argv[i]; + switch (arg) { + case "--dry-run": + args.dryRun = true; + break; + case "--only": + i += 1; + args.only = argv[i] ?? null; + break; + case "--snapshots": + i += 1; + args.root = argv[i] ?? args.root; + break; + case "--user-agent": + i += 1; + args.userAgent = argv[i] ?? args.userAgent; + break; + case "--no-feed": + args.rebuild = false; + break; + case "--help": + case "-h": + args.help = true; + break; + default: + if (arg !== undefined && arg.startsWith("-")) { + throw new Error(`unknown flag '${arg}'`); + } + break; + } + } + + return args; +} + +const USAGE = `usage: bun run refresh [--dry-run] [--only ] [--snapshots ] + [--user-agent ] [--no-feed] + + --dry-run report what each source would do; no requests, no writes + --only refresh a single source (${ADAPTERS.map((a) => a.id).join(", ")}) + --snapshots snapshot cache directory (default: snapshots, env SNAPSHOT_DIR) + --user-agent override the User-Agent (env REFRESH_USER_AGENT) + --no-feed skip regenerating public/data/events.v1.json`; + +async function main(): Promise { + let args: Args; + try { + args = parseArgs(Bun.argv.slice(2)); + } catch (error) { + console.error(String(error)); + console.error(USAGE); + return 2; + } + + if (args.help) { + console.log(USAGE); + return 0; + } + + if (args.only !== null && adapterById(args.only) === undefined) { + console.error(`unknown source '${args.only}'`); + console.error(USAGE); + return 2; + } + + const store = new SnapshotStore(args.root); + const robots = new RobotsCache({ + userAgent: args.userAgent, + fetchImpl: (input, init) => fetch(input, init), + }); + + console.log( + `refresh: ${args.only ?? `${ADAPTERS.length} sources`}${args.dryRun ? " (dry run)" : ""}`, + ); + console.log(` user-agent: ${args.userAgent}`); + console.log(` snapshots: ${args.root}`); + console.log(` interval: ${SIX_HOURS_MS / 3_600_000}h minimum per source\n`); + + const summary = await runRefresh({ + adapters: ADAPTERS, + store, + robots, + fetchImpl: (input, init) => fetch(input, init), + userAgent: args.userAgent, + now: () => new Date(), + dryRun: args.dryRun, + only: args.only, + timeoutMs: 20_000, + log: (line) => console.log(line), + rebuildFeed: args.dryRun || !args.rebuild ? null : rebuildFeedViaScript, + }); + + console.log( + `\n${summary.changed} changed, ${summary.confirmed}/${summary.attempted} confirmed, ${summary.warnings.length} warnings`, + ); + for (const warning of summary.warnings) console.warn(` ! ${warning}`); + + // The workflow reads this line to decide whether to commit; `git status` is + // the authority, but this makes a skipped commit legible in the log. + console.log(`changed=${summary.changed}`); + + if (summary.hardFailure !== null) { + console.error(`\nrefresh failed: ${summary.hardFailure}`); + return 1; + } + + return 0; +} + +if (import.meta.main) { + process.exit(await main()); +} diff --git a/snapshots/README.md b/snapshots/README.md new file mode 100644 index 0000000..3f9b9f3 --- /dev/null +++ b/snapshots/README.md @@ -0,0 +1,25 @@ +# Snapshots + +Raw pages, exactly as fetched. `scripts/refresh-sources.ts` writes them; nothing else fetches. + +``` +.html the body verbatim — tracked +.meta.json hash, size, ETag, Last-Modified, when the bytes last changed — tracked +.state.json when we last checked, and failure streak — gitignored +``` + +Three reasons this is committed rather than cached: + +- **Re-parsing never re-fetches.** Iterating on a parser reads these files, not the wikis + (CLAUDE.md § Scraping conduct). +- **A refresh is reviewable.** The commit diff is the page diff, so "an event vanished" is a + question you can answer from git rather than from a wiki that has since changed again. +- **The build stays offline.** `bun run build:feed` parses whichever of these exists and falls back + to `fixtures/` otherwise, so a clean checkout and the container build work with no network. + +The `.state.json` files are the exception: they change every cycle whether or not a page did, and +committing them would mean a commit per run saying nothing happened. CI keeps them in the actions +cache instead. + +Fixtures are not the same thing. A fixture is pinned to a date and kept forever as the regression +test for a page shape; a snapshot is the current page and is overwritten each time it changes. diff --git a/src/ingest/robots.ts b/src/ingest/robots.ts new file mode 100644 index 0000000..0e7cd71 --- /dev/null +++ b/src/ingest/robots.ts @@ -0,0 +1,374 @@ +/** + * robots.txt: parsing, matching, and a per-host cache. + * + * Sources are community wikis and this project's standing rule is to behave as + * a guest would (CLAUDE.md § Scraping conduct). That starts with actually + * reading robots.txt rather than assuming a path is fair game. + * + * Parsing is a pure function over text, deliberately separated from fetching, + * so every matching rule below is unit-testable offline. Only `RobotsCache` + * touches the network, and it takes its `fetch` by injection. + * + * Follows RFC 9309: user-agent groups, Allow/Disallow with `*` and `$` + * wildcards, longest-match-wins with Allow winning a tie, and `*` as the + * fallback group used only when no named group matches. + */ + +export interface RobotsRule { + /** true for `Allow:`, false for `Disallow:`. */ + readonly allow: boolean; + /** The raw path pattern; may contain `*` and a trailing `$`. */ + readonly pattern: string; +} + +export interface RobotsGroup { + /** Lowercased user-agent tokens this group applies to. `*` is the fallback. */ + readonly agents: readonly string[]; + readonly rules: readonly RobotsRule[]; + readonly crawlDelaySeconds: number | null; +} + +export interface RobotsTxt { + readonly groups: readonly RobotsGroup[]; + readonly sitemaps: readonly string[]; +} + +/** A robots.txt that restricts nothing — what an absent file means. */ +export const ALLOW_ALL: RobotsTxt = { groups: [], sitemaps: [] }; + +export type FetchLike = ( + input: string, + init?: RequestInit, +) => Promise; + +interface MutableGroup { + agents: string[]; + rules: RobotsRule[]; + crawlDelaySeconds: number | null; +} + +/** + * Parse robots.txt text. + * + * Unknown directives are ignored rather than treated as errors — a file we do + * not fully understand must still yield the rules we do understand. + */ +export function parseRobots(text: string): RobotsTxt { + const groups: MutableGroup[] = []; + const sitemaps: string[] = []; + + let current: MutableGroup | null = null; + // Consecutive `User-agent:` lines share one group; the first rule line after + // them closes the agent list, so the next `User-agent:` starts a new group. + let acceptingAgents = false; + + for (const rawLine of text.split(/\r?\n/)) { + const line = stripComment(rawLine).trim(); + if (line === "") continue; + + const colon = line.indexOf(":"); + if (colon === -1) continue; + + const key = line.slice(0, colon).trim().toLowerCase(); + const value = line.slice(colon + 1).trim(); + + switch (key) { + case "user-agent": { + if (value === "") break; + if (current === null || !acceptingAgents) { + current = { agents: [], rules: [], crawlDelaySeconds: null }; + groups.push(current); + acceptingAgents = true; + } + current.agents.push(value.toLowerCase()); + break; + } + case "allow": + case "disallow": { + if (current === null) break; + acceptingAgents = false; + // `Disallow:` with an empty value is the documented way to say + // "nothing is disallowed", so it must not become a match-everything + // rule. An empty `Allow:` is equally inert. + if (value === "") break; + current.rules.push({ allow: key === "allow", pattern: value }); + break; + } + case "crawl-delay": { + if (current === null) break; + acceptingAgents = false; + const seconds = Number(value); + if (Number.isFinite(seconds) && seconds >= 0) { + current.crawlDelaySeconds = seconds; + } + break; + } + case "sitemap": { + if (value !== "") sitemaps.push(value); + break; + } + default: + break; + } + } + + return { + groups: groups.map((g) => ({ + agents: g.agents, + rules: g.rules, + crawlDelaySeconds: g.crawlDelaySeconds, + })), + sitemaps, + }; +} + +function stripComment(line: string): string { + const hash = line.indexOf("#"); + return hash === -1 ? line : line.slice(0, hash); +} + +/** + * The product token of a User-Agent header. + * + * `"gacha-event-tracker/1.0 (+https://example.test)"` → `"gacha-event-tracker"`. + */ +export function agentToken(userAgent: string): string { + const first = userAgent.trim().split(/[\s/]/, 1)[0] ?? ""; + return first.toLowerCase(); +} + +/** + * The group that applies to a user agent, with every group naming the same + * agent merged, as RFC 9309 requires. + * + * A named group beats `*` outright: a site that disallows everything for `*` + * but names us explicitly is telling us we may fetch. Longest agent name wins + * among several matches, so `googlebot-news` beats `googlebot`. + */ +export function groupFor( + robots: RobotsTxt, + userAgent: string, +): RobotsGroup | null { + const token = agentToken(userAgent); + const full = userAgent.toLowerCase(); + + let bestName: string | null = null; + for (const group of robots.groups) { + for (const agent of group.agents) { + if (agent === "*") continue; + // Match on the product token first (the spec's rule); fall back to a + // substring of the whole header so a group naming "gptbot" still binds a + // header of "Mozilla/5.0 (compatible; GPTBot/1.2)". Erring towards + // matching means erring towards obeying more rules, not fewer. + const hit = + token === agent || token.startsWith(agent) || full.includes(agent); + if (!hit) continue; + if (bestName === null || agent.length > bestName.length) bestName = agent; + } + } + + const name = bestName ?? "*"; + const matching = robots.groups.filter((g) => g.agents.includes(name)); + if (matching.length === 0) return null; + + return { + agents: [name], + rules: matching.flatMap((g) => g.rules), + crawlDelaySeconds: + matching.reduce( + (acc, g) => + g.crawlDelaySeconds === null + ? acc + : Math.max(acc ?? 0, g.crawlDelaySeconds), + null, + ) ?? null, + }; +} + +/** Does a robots path pattern match this path? Supports `*` and a final `$`. */ +export function patternMatches(pattern: string, path: string): boolean { + if (pattern === "") return false; + + const anchored = pattern.endsWith("$"); + const body = anchored ? pattern.slice(0, -1) : pattern; + + let regex = ""; + for (const char of body) { + if (char === "*") { + regex += "[\\s\\S]*"; + } else { + regex += char.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); + } + } + + return new RegExp(`^${regex}${anchored ? "$" : ""}`).test(path); +} + +/** + * May `userAgent` fetch `path`? + * + * `path` is the request target — pathname plus query string, e.g. `/wiki/Event`. + * Longest matching pattern wins; a tie goes to Allow; no match means allowed. + */ +export function isAllowed( + robots: RobotsTxt, + userAgent: string, + path: string, +): boolean { + const group = groupFor(robots, userAgent); + if (group === null) return true; + + const target = path.startsWith("/") ? path : `/${path}`; + + let bestLength = -1; + let allowed = true; + for (const rule of group.rules) { + if (!patternMatches(rule.pattern, target)) continue; + const length = rule.pattern.length; + if (length > bestLength || (length === bestLength && rule.allow)) { + bestLength = length; + allowed = rule.allow; + } + } + + return allowed; +} + +/** The crawl delay this agent should honour, in ms, if the file states one. */ +export function crawlDelayMs( + robots: RobotsTxt, + userAgent: string, +): number | null { + const group = groupFor(robots, userAgent); + if (group === null || group.crawlDelaySeconds === null) return null; + return Math.round(group.crawlDelaySeconds * 1000); +} + +/** The path-and-query a robots rule is matched against. */ +export function requestTarget(url: string): string { + const parsed = new URL(url); + return `${parsed.pathname}${parsed.search}`; +} + +export interface RobotsDecision { + readonly allowed: boolean; + /** Human-readable why, for the run log. */ + readonly reason: string; + readonly crawlDelayMs: number | null; +} + +export interface RobotsCacheOptions { + userAgent: string; + fetchImpl: FetchLike; + /** How long a parsed robots.txt stays good. Defaults to 24h, per docs. */ + ttlMs?: number; + now?: () => number; + timeoutMs?: number; +} + +interface CacheEntry { + robots: RobotsTxt; + /** False when robots.txt could not be read; the host is then off limits. */ + usable: boolean; + reason: string; + at: number; +} + +const DAY_MS = 24 * 60 * 60 * 1000; + +/** + * One robots.txt fetch per host per run (cached 24h), reused by every source on + * that host — six Game8 adapters must not mean six robots requests. + * + * Fails closed. A 5xx, a timeout or a network error means we do not know what + * the site permits, and "unknown" is not permission. + */ +export class RobotsCache { + private readonly entries = new Map(); + private readonly userAgent: string; + private readonly fetchImpl: FetchLike; + private readonly ttlMs: number; + private readonly nowMs: () => number; + private readonly timeoutMs: number; + + constructor(options: RobotsCacheOptions) { + this.userAgent = options.userAgent; + this.fetchImpl = options.fetchImpl; + this.ttlMs = options.ttlMs ?? DAY_MS; + this.nowMs = options.now ?? (() => Date.now()); + this.timeoutMs = options.timeoutMs ?? 20_000; + } + + /** Number of robots.txt requests made, for tests and the run log. */ + fetches = 0; + + async allows(url: string): Promise { + const origin = new URL(url).origin; + const entry = await this.entryFor(origin); + + if (!entry.usable) { + return { allowed: false, reason: entry.reason, crawlDelayMs: null }; + } + + const allowed = isAllowed(entry.robots, this.userAgent, requestTarget(url)); + return { + allowed, + reason: allowed ? entry.reason : `disallowed by ${origin}/robots.txt`, + crawlDelayMs: crawlDelayMs(entry.robots, this.userAgent), + }; + } + + private async entryFor(origin: string): Promise { + const cached = this.entries.get(origin); + if (cached !== undefined && this.nowMs() - cached.at < this.ttlMs) { + return cached; + } + + const entry = await this.load(origin); + this.entries.set(origin, entry); + return entry; + } + + private async load(origin: string): Promise { + const at = this.nowMs(); + this.fetches += 1; + + let response: Response; + try { + response = await this.fetchImpl(`${origin}/robots.txt`, { + headers: { "User-Agent": this.userAgent, Accept: "text/plain" }, + signal: AbortSignal.timeout(this.timeoutMs), + redirect: "follow", + }); + } catch (error) { + return { + robots: ALLOW_ALL, + usable: false, + reason: `robots.txt unreachable (${String(error)})`, + at, + }; + } + + if (response.status === 404 || response.status === 410) { + // No robots.txt is the site saying nothing, which means no restrictions. + return { + robots: ALLOW_ALL, + usable: true, + reason: "no robots.txt", + at, + }; + } + + if (response.status >= 400) { + return { + robots: ALLOW_ALL, + usable: false, + reason: `robots.txt returned ${response.status}`, + at, + }; + } + + const text = await response.text(); + return { robots: parseRobots(text), usable: true, reason: "robots.txt ok", at }; + } +} diff --git a/src/ingest/snapshots.ts b/src/ingest/snapshots.ts new file mode 100644 index 0000000..2b209a2 --- /dev/null +++ b/src/ingest/snapshots.ts @@ -0,0 +1,261 @@ +/** + * The raw snapshot cache. + * + * Every fetched page is stored verbatim on disk so that re-parsing — which is + * the thing we actually iterate on — never costs the source another request + * (CLAUDE.md § Scraping conduct). A snapshot plus its metadata is also what + * makes a conditional request possible on the next cycle: we keep the ETag and + * Last-Modified the server gave us and hand them back. + * + * Three files per source, and the split matters: + * + * /.html the body, exactly as served + * /.meta.json durable facts: hash, validators, when it changed + * /.state.json volatile run bookkeeping: when we last checked + * + * `.state.json` is separated (and gitignored) so that a refresh which confirms + * "nothing changed" leaves a clean working tree. If check timestamps lived in + * the metadata, every cycle would produce a commit that says nothing, and + * "commit only when something changed" would be unenforceable. + */ +import { mkdir, readdir, rm, writeFile } from "node:fs/promises"; +import { join } from "node:path"; + +export interface SnapshotMeta { + sourceId: string; + url: string; + /** sha256 of the body, hex. The parse stage skips work when it is unchanged. */ + contentHash: string; + bytes: number; + etag: string | null; + lastModified: string | null; + /** ISO timestamp of the fetch that last produced *different* bytes. */ + contentChangedAt: string; + /** Events the adapter yielded from this body, for drop detection. */ + eventCount: number | null; +} + +export interface SnapshotState { + sourceId: string; + /** Last attempt of any kind — this is what the minimum interval reads. */ + lastCheckedAt: string | null; + /** Last time the server confirmed the body (200 or 304). */ + lastConfirmedAt: string | null; + lastStatus: number | null; + consecutiveFailures: number; +} + +export interface Snapshot { + meta: SnapshotMeta; + state: SnapshotState; + html: string; +} + +export interface SaveInput { + url: string; + html: string; + etag: string | null; + lastModified: string | null; + /** ISO timestamp of this fetch. */ + at: string; + eventCount: number | null; +} + +export interface SaveResult { + changed: boolean; + meta: SnapshotMeta; +} + +export function hashBody(html: string): string { + return new Bun.CryptoHasher("sha256").update(html).digest("hex"); +} + +function emptyState(sourceId: string): SnapshotState { + return { + sourceId, + lastCheckedAt: null, + lastConfirmedAt: null, + lastStatus: null, + consecutiveFailures: 0, + }; +} + +export class SnapshotStore { + constructor(readonly root: string = "snapshots") {} + + bodyPath(sourceId: string): string { + return join(this.root, `${sourceId}.html`); + } + + metaPath(sourceId: string): string { + return join(this.root, `${sourceId}.meta.json`); + } + + statePath(sourceId: string): string { + return join(this.root, `${sourceId}.state.json`); + } + + async readMeta(sourceId: string): Promise { + const file = Bun.file(this.metaPath(sourceId)); + if (!(await file.exists())) return null; + try { + return (await file.json()) as SnapshotMeta; + } catch { + // A truncated metadata file must not take the run down; treat it as no + // cache, which costs one full fetch and self-heals. + return null; + } + } + + async readState(sourceId: string): Promise { + const file = Bun.file(this.statePath(sourceId)); + if (!(await file.exists())) return emptyState(sourceId); + try { + return { ...emptyState(sourceId), ...((await file.json()) as SnapshotState) }; + } catch { + return emptyState(sourceId); + } + } + + /** Body plus metadata, or null when this source has never been fetched. */ + async read(sourceId: string): Promise { + const meta = await this.readMeta(sourceId); + if (meta === null) return null; + + const body = Bun.file(this.bodyPath(sourceId)); + if (!(await body.exists())) return null; + + return { meta, state: await this.readState(sourceId), html: await body.text() }; + } + + /** Sources with a stored snapshot, by id. */ + async list(): Promise { + let names: string[]; + try { + names = await readdir(this.root); + } catch { + return []; + } + return names + .filter((n) => n.endsWith(".meta.json")) + .map((n) => n.slice(0, -".meta.json".length)) + .sort(); + } + + /** + * Headers for the next request. An empty object is correct for a source we + * have never seen — there is nothing to be conditional about. + */ + conditionalHeaders(meta: SnapshotMeta | null): Record { + const headers: Record = {}; + if (meta === null) return headers; + if (meta.etag !== null && meta.etag !== "") { + headers["If-None-Match"] = meta.etag; + } + if (meta.lastModified !== null && meta.lastModified !== "") { + headers["If-Modified-Since"] = meta.lastModified; + } + return headers; + } + + /** + * Has enough time passed to fetch this source again? + * + * The floor is six hours per source (CLAUDE.md). A source we have never + * checked is always due. + */ + isDue(state: SnapshotState, nowMs: number, minIntervalMs: number): boolean { + if (state.lastCheckedAt === null) return true; + const last = Date.parse(state.lastCheckedAt); + if (Number.isNaN(last)) return true; + return nowMs - last >= minIntervalMs; + } + + /** When this source may next be fetched, in epoch ms. */ + dueAt(state: SnapshotState, minIntervalMs: number): number { + if (state.lastCheckedAt === null) return 0; + const last = Date.parse(state.lastCheckedAt); + return Number.isNaN(last) ? 0 : last + minIntervalMs; + } + + /** + * Store a fetched body. + * + * Identical bytes are a no-op on disk: the metadata keeps its original + * `contentChangedAt` and validators, so an unchanged source produces no diff + * for the workflow to commit. + */ + async save(sourceId: string, input: SaveInput): Promise { + const contentHash = hashBody(input.html); + const previous = await this.readMeta(sourceId); + const bodyExists = await Bun.file(this.bodyPath(sourceId)).exists(); + const changed = + previous === null || previous.contentHash !== contentHash || !bodyExists; + + if (!changed && previous !== null) { + return { changed: false, meta: previous }; + } + + const meta: SnapshotMeta = { + sourceId, + url: input.url, + contentHash, + bytes: Buffer.byteLength(input.html), + etag: input.etag, + lastModified: input.lastModified, + contentChangedAt: input.at, + eventCount: input.eventCount, + }; + + await mkdir(this.root, { recursive: true }); + await writeFile(this.bodyPath(sourceId), input.html); + await writeFile(this.metaPath(sourceId), `${JSON.stringify(meta, null, 2)}\n`); + + return { changed: true, meta }; + } + + /** Record an attempt: success, 304 or failure. Writes only state. */ + async recordCheck( + sourceId: string, + check: { at: string; status: number | null; ok: boolean }, + ): Promise { + const previous = await this.readState(sourceId); + const state: SnapshotState = { + sourceId, + lastCheckedAt: check.at, + lastConfirmedAt: check.ok ? check.at : previous.lastConfirmedAt, + lastStatus: check.status, + consecutiveFailures: check.ok ? 0 : previous.consecutiveFailures + 1, + }; + + await mkdir(this.root, { recursive: true }); + await writeFile(this.statePath(sourceId), `${JSON.stringify(state, null, 2)}\n`); + return state; + } + + /** Remove one source's cache entirely. Used by tests and by hand. */ + async forget(sourceId: string): Promise { + for (const path of [ + this.bodyPath(sourceId), + this.metaPath(sourceId), + this.statePath(sourceId), + ]) { + await rm(path, { force: true }); + } + } +} + +/** + * How fresh a snapshot demonstrably is. + * + * The last time the server confirmed the body, when we know it; otherwise the + * last time the content changed. Never later than reality — a freshness badge + * that overstates is worse than one that lags. + */ +export function freshnessAt(snapshot: Snapshot): string { + const confirmed = snapshot.state.lastConfirmedAt; + if (confirmed === null) return snapshot.meta.contentChangedAt; + return Date.parse(confirmed) > Date.parse(snapshot.meta.contentChangedAt) + ? confirmed + : snapshot.meta.contentChangedAt; +} diff --git a/test/refresh.test.ts b/test/refresh.test.ts new file mode 100644 index 0000000..3e0e86c --- /dev/null +++ b/test/refresh.test.ts @@ -0,0 +1,439 @@ +import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import { mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { + parseArgs, + runRefresh, + type RefreshOptions, + type RobotsGate, +} from "../scripts/refresh-sources.ts"; +import type { Adapter, ParseContext } from "../src/ingest/adapters/types.ts"; +import { SnapshotStore } from "../src/ingest/snapshots.ts"; +import type { GachaEvent } from "../src/shared/schema.ts"; + +const SIX_HOURS_MS = 6 * 60 * 60 * 1000; +const NOW = new Date("2026-08-15T12:00:00.000Z"); +const UA = "gacha-event-tracker/1.0 (+https://example.test)"; + +let root: string; +let store: SnapshotStore; + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), "event-clock-refresh-")); + store = new SnapshotStore(root); +}); + +afterEach(async () => { + await rm(root, { recursive: true, force: true }); +}); + +/** + * A stand-in adapter whose "parser" counts `` tags, so a test can make a + * body parse well, badly, or not at all without touching a real parser. + */ +function adapter(overrides: Partial = {}): Adapter { + return { + id: "genshin-game8-events", + game: "genshin", + url: "https://game8.co/games/Genshin-Impact/archives/301601", + parserId: "game8", + minIntervalMs: SIX_HOURS_MS, + priority: 0, + parse(html: string, _ctx: ParseContext): GachaEvent[] { + if (html.includes("")) throw new Error("template not recognised"); + const count = html.match(//g)?.length ?? 0; + return Array.from({ length: count }) as GachaEvent[]; + }, + ...overrides, + }; +} + +const ALLOW_ALL: RobotsGate = { + allows: async () => ({ allowed: true, reason: "robots.txt ok" }), +}; + +interface Call { + url: string; + headers: Record; +} + +function options( + over: Partial & { responder?: (call: Call) => Response }, +): { opts: RefreshOptions; calls: Call[]; rebuilds: { count: number } } { + const calls: Call[] = []; + const rebuilds = { count: 0 }; + const responder = + over.responder ?? (() => new Response("")); + + const opts: RefreshOptions = { + adapters: [adapter()], + store, + robots: ALLOW_ALL, + fetchImpl: async (url, init) => { + const headers: Record = {}; + for (const [k, v] of Object.entries( + (init?.headers ?? {}) as Record, + )) { + headers[k] = v; + } + const call = { url, headers }; + calls.push(call); + return responder(call); + }, + userAgent: UA, + now: () => NOW, + dryRun: false, + only: null, + timeoutMs: 1000, + log: () => {}, + rebuildFeed: async () => { + rebuilds.count += 1; + }, + ...over, + }; + + return { opts, calls, rebuilds }; +} + +async function seed(html: string, at: string, eventCount: number | null) { + await store.save("genshin-game8-events", { + url: "https://game8.co/games/Genshin-Impact/archives/301601", + html, + etag: 'W/"v1"', + lastModified: "Fri, 14 Aug 2026 09:00:00 GMT", + at, + eventCount, + }); +} + +describe("a normal cycle", () => { + test("fetches once, stores the body, rebuilds the feed", async () => { + const { opts, calls, rebuilds } = options({}); + const summary = await runRefresh(opts); + + expect(calls).toHaveLength(1); + expect(summary.outcomes[0]?.result).toBe("fetched"); + expect(summary.changed).toBe(1); + expect(summary.hardFailure).toBeNull(); + expect(rebuilds.count).toBe(1); + + const snapshot = await store.read("genshin-game8-events"); + expect(snapshot?.html).toBe(""); + expect(snapshot?.meta.eventCount).toBe(1); + expect(snapshot?.state.lastConfirmedAt).toBe(NOW.toISOString()); + }); + + test("identifies itself with a contact URL", async () => { + const { opts, calls } = options({}); + await runRefresh(opts); + expect(calls[0]?.headers["User-Agent"]).toBe(UA); + expect(calls[0]?.headers["User-Agent"]).toContain("+https://"); + }); + + test("sends the validators it was given last time", async () => { + await seed("", "2026-08-01T00:00:00.000Z", 1); + const { opts, calls } = options({}); + await runRefresh(opts); + + expect(calls[0]?.headers["If-None-Match"]).toBe('W/"v1"'); + expect(calls[0]?.headers["If-Modified-Since"]).toBe( + "Fri, 14 Aug 2026 09:00:00 GMT", + ); + }); + + test("304 reuses the cached snapshot and changes nothing", async () => { + await seed("", "2026-08-01T00:00:00.000Z", 1); + const { opts, rebuilds } = options({ + responder: () => new Response(null, { status: 304 }), + }); + const summary = await runRefresh(opts); + + expect(summary.outcomes[0]?.result).toBe("unchanged"); + expect(summary.changed).toBe(0); + expect(rebuilds.count).toBe(0); + + const snapshot = await store.read("genshin-game8-events"); + expect(snapshot?.html).toBe(""); + expect(snapshot?.meta.contentChangedAt).toBe("2026-08-01T00:00:00.000Z"); + expect(snapshot?.state.lastConfirmedAt).toBe(NOW.toISOString()); + }); + + test("a 200 with identical bytes is not a change either", async () => { + await seed("", "2026-08-01T00:00:00.000Z", 1); + const { opts, rebuilds } = options({}); + const summary = await runRefresh(opts); + + expect(summary.outcomes[0]?.result).toBe("unchanged"); + expect(summary.changed).toBe(0); + expect(rebuilds.count).toBe(0); + }); +}); + +describe("one request per source per six hours", () => { + test("skips a source checked less than six hours ago", async () => { + await store.recordCheck("genshin-game8-events", { + at: new Date(NOW.getTime() - SIX_HOURS_MS + 1000).toISOString(), + status: 200, + ok: true, + }); + + const { opts, calls } = options({}); + const summary = await runRefresh(opts); + + expect(calls).toHaveLength(0); + expect(summary.outcomes[0]?.result).toBe("skipped_interval"); + expect(summary.attempted).toBe(0); + expect(summary.hardFailure).toBeNull(); + }); + + test("fetches again once the interval has elapsed", async () => { + await store.recordCheck("genshin-game8-events", { + at: new Date(NOW.getTime() - SIX_HOURS_MS).toISOString(), + status: 200, + ok: true, + }); + + const { opts, calls } = options({}); + await runRefresh(opts); + expect(calls).toHaveLength(1); + }); + + test("never retries a failure inside the same cycle", async () => { + const { opts, calls } = options({ + responder: () => new Response("nope", { status: 500 }), + }); + const summary = await runRefresh(opts); + + expect(calls).toHaveLength(1); + expect(summary.outcomes[0]?.result).toBe("failed"); + }); +}); + +describe("robots", () => { + test("does not fetch a source robots.txt disallows", async () => { + const { opts, calls } = options({ + robots: { + allows: async () => ({ allowed: false, reason: "disallowed by robots" }), + }, + }); + const summary = await runRefresh(opts); + + expect(calls).toHaveLength(0); + expect(summary.outcomes[0]?.result).toBe("skipped_robots"); + expect(summary.warnings).toHaveLength(1); + }); + + test("one source blocked is a warning; all of them is a failure", async () => { + const blocked = { + allows: async (url: string) => ({ + allowed: !url.includes("Genshin"), + reason: "disallowed by robots", + }), + }; + const two = [ + adapter(), + adapter({ + id: "nte-game8-events", + game: "nte", + url: "https://game8.co/games/Neverness-to-Everness/archives/592073", + }), + ]; + + const partial = await runRefresh(options({ adapters: two, robots: blocked }).opts); + expect(partial.hardFailure).toBeNull(); + expect(partial.warnings).toHaveLength(1); + + // Fresh ids: the partial run above already checked one of the pair, and a + // source checked minutes ago is skipped for the interval, not for robots. + const all = await runRefresh( + options({ + adapters: [ + adapter({ id: "hsr-game8-events", game: "hsr", url: "https://game8.co/a" }), + adapter({ id: "zzz-game8-events", game: "zzz", url: "https://game8.co/b" }), + ], + robots: { + allows: async () => ({ allowed: false, reason: "disallowed by robots" }), + }, + }).opts, + ); + expect(all.hardFailure).toContain("blocked all 2 sources"); + }); + + test("robots is consulted before the page is requested", async () => { + const order: string[] = []; + const { opts } = options({ + robots: { + allows: async () => { + order.push("robots"); + return { allowed: true, reason: "ok" }; + }, + }, + responder: () => { + order.push("page"); + return new Response(""); + }, + }); + await runRefresh(opts); + expect(order).toEqual(["robots", "page"]); + }); +}); + +describe("a source being down never blanks the feed", () => { + test("an unreachable source is a warning, not a failure", async () => { + await seed("", "2026-08-01T00:00:00.000Z", 1); + const { opts } = options({ + adapters: [ + adapter(), + adapter({ + id: "nte-game8-events", + game: "nte", + url: "https://game8.co/games/Neverness-to-Everness/archives/592073", + }), + ], + responder: (call) => { + if (call.url.includes("Genshin")) throw new Error("ETIMEDOUT"); + return new Response(""); + }, + }); + + const summary = await runRefresh(opts); + + expect(summary.outcomes[0]?.result).toBe("failed"); + expect(summary.outcomes[1]?.result).toBe("fetched"); + expect(summary.warnings).toHaveLength(1); + expect(summary.hardFailure).toBeNull(); + // The old snapshot is untouched, so the feed keeps this game's events. + expect((await store.read("genshin-game8-events"))?.html).toBe( + "", + ); + }); + + test("every source failing is a hard failure", async () => { + const { opts, rebuilds } = options({ + adapters: [adapter(), adapter({ id: "nte-game8-events", game: "nte" })], + responder: () => new Response("", { status: 503 }), + }); + const summary = await runRefresh(opts); + + expect(summary.hardFailure).toContain("all 2 attempted sources failed"); + expect(rebuilds.count).toBe(0); + }); + + test("a body that no longer parses keeps the previous snapshot", async () => { + await seed("", "2026-08-01T00:00:00.000Z", 1); + const { opts } = options({ + responder: () => new Response(""), + }); + const summary = await runRefresh(opts); + + expect(summary.outcomes[0]?.result).toBe("rejected"); + expect(summary.warnings[0]).toContain("did not parse"); + expect((await store.read("genshin-game8-events"))?.html).toBe( + "", + ); + }); + + test("a body that suddenly yields no events keeps the previous snapshot", async () => { + await seed("", "2026-08-01T00:00:00.000Z", 1); + const { opts } = options({ + responder: () => new Response("redesigned"), + }); + const summary = await runRefresh(opts); + + expect(summary.outcomes[0]?.result).toBe("rejected"); + expect(summary.warnings[0]).toContain("0 events"); + expect((await store.read("genshin-game8-events"))?.meta.eventCount).toBe(1); + }); + + test("a first fetch that yields nothing is not stored either", async () => { + // With no snapshot yet, storing an empty parse would make build-feed prefer + // it over the checked-in fixture and quietly empty that game's calendar. + const { opts } = options({ + responder: () => new Response("redesigned"), + }); + const summary = await runRefresh(opts); + + expect(summary.outcomes[0]?.result).toBe("rejected"); + expect(summary.changed).toBe(0); + expect(await store.read("genshin-game8-events")).toBeNull(); + }); + + test("a steep drop is stored but flagged", async () => { + await seed("" + "".repeat(10) + "", "2026-08-01T00:00:00.000Z", 10); + const { opts } = options({ + responder: () => new Response(""), + }); + const summary = await runRefresh(opts); + + expect(summary.outcomes[0]?.result).toBe("fetched"); + expect(summary.outcomes[0]?.note).toContain("down from 10"); + }); + + test("a feed that will not rebuild fails the run", async () => { + const { opts } = options({ + rebuildFeed: async () => { + throw new Error("build-feed exited 1"); + }, + }); + const summary = await runRefresh(opts); + expect(summary.hardFailure).toContain("feed rebuild failed"); + }); +}); + +describe("flags", () => { + test("--dry-run makes no requests and writes nothing", async () => { + const { opts, calls, rebuilds } = options({ dryRun: true }); + const summary = await runRefresh(opts); + + expect(calls).toHaveLength(0); + expect(rebuilds.count).toBe(0); + expect(summary.outcomes[0]?.result).toBe("planned"); + expect(summary.outcomes[0]?.note).toContain("would GET"); + expect(await store.read("genshin-game8-events")).toBeNull(); + expect(await store.readState("genshin-game8-events")).toMatchObject({ + lastCheckedAt: null, + }); + }); + + test("--only refreshes one source", async () => { + const { opts, calls } = options({ + adapters: [adapter(), adapter({ id: "nte-game8-events", game: "nte" })], + only: "nte-game8-events", + }); + const summary = await runRefresh(opts); + + expect(calls).toHaveLength(1); + expect(summary.outcomes).toHaveLength(1); + expect(summary.outcomes[0]?.sourceId).toBe("nte-game8-events"); + }); + + test("--only with an unknown id is a hard failure", async () => { + const { opts, calls } = options({ only: "does-not-exist" }); + const summary = await runRefresh(opts); + + expect(calls).toHaveLength(0); + expect(summary.hardFailure).toContain("unknown source"); + }); + + test("parseArgs reads the flags", () => { + const args = parseArgs([ + "--dry-run", + "--only", + "nte-game8-events", + "--snapshots", + "/tmp/x", + "--no-feed", + ]); + expect(args).toMatchObject({ + dryRun: true, + only: "nte-game8-events", + root: "/tmp/x", + rebuild: false, + }); + }); + + test("parseArgs rejects an unknown flag rather than ignoring it", () => { + expect(() => parseArgs(["--force"])).toThrow("unknown flag"); + }); +}); diff --git a/test/robots.test.ts b/test/robots.test.ts new file mode 100644 index 0000000..1038b89 --- /dev/null +++ b/test/robots.test.ts @@ -0,0 +1,260 @@ +import { describe, expect, test } from "bun:test"; +import { + agentToken, + crawlDelayMs, + groupFor, + isAllowed, + parseRobots, + patternMatches, + requestTarget, + RobotsCache, +} from "../src/ingest/robots.ts"; + +const UA = "gacha-event-tracker/1.0 (+https://example.test/contact)"; + +describe("parseRobots", () => { + test("groups consecutive user-agent lines together", () => { + const robots = parseRobots(` +User-agent: alpha +User-agent: beta +Disallow: /private + +User-agent: * +Disallow: / +`); + expect(robots.groups).toHaveLength(2); + expect(robots.groups[0]?.agents).toEqual(["alpha", "beta"]); + expect(robots.groups[1]?.agents).toEqual(["*"]); + }); + + test("ignores comments, blank lines and unknown directives", () => { + const robots = parseRobots( + "# a comment\r\nUser-agent: * # trailing\r\nHost: example.test\r\nDisallow: /x\r\nSitemap: https://example.test/sitemap.xml\r\n", + ); + expect(robots.groups[0]?.rules).toEqual([{ allow: false, pattern: "/x" }]); + expect(robots.sitemaps).toEqual(["https://example.test/sitemap.xml"]); + }); + + test("an empty Disallow restricts nothing", () => { + const robots = parseRobots("User-agent: *\nDisallow:\n"); + expect(robots.groups[0]?.rules).toEqual([]); + expect(isAllowed(robots, UA, "/anything")).toBe(true); + }); + + test("reads crawl-delay", () => { + const robots = parseRobots("User-agent: *\nCrawl-delay: 10\nDisallow: /x\n"); + expect(crawlDelayMs(robots, UA)).toBe(10_000); + }); +}); + +describe("group selection", () => { + test("takes the product token out of a full User-Agent header", () => { + expect(agentToken(UA)).toBe("gacha-event-tracker"); + }); + + test("a named group beats the wildcard group", () => { + const robots = parseRobots(` +User-agent: * +Disallow: / + +User-agent: gacha-event-tracker +Disallow: /admin +`); + expect(isAllowed(robots, UA, "/games/Genshin-Impact/archives/301601")).toBe( + true, + ); + expect(isAllowed(robots, UA, "/admin/panel")).toBe(false); + }); + + test("falls back to the wildcard group when nothing names us", () => { + const robots = parseRobots("User-agent: *\nDisallow: /games/\n"); + expect(isAllowed(robots, UA, "/games/x")).toBe(false); + }); + + test("no applicable group at all means allowed", () => { + const robots = parseRobots("User-agent: gptbot\nDisallow: /\n"); + expect(groupFor(robots, UA)).toBeNull(); + expect(isAllowed(robots, UA, "/games/x")).toBe(true); + }); + + test("the longest matching agent name wins", () => { + const robots = parseRobots(` +User-agent: googlebot +Disallow: / + +User-agent: googlebot-news +Allow: / +`); + expect(isAllowed(robots, "Googlebot-News/1.0", "/anything")).toBe(true); + expect(isAllowed(robots, "Googlebot/2.1", "/anything")).toBe(false); + }); + + test("merges rules from several groups naming the same agent", () => { + const robots = parseRobots(` +User-agent: gacha-event-tracker +Disallow: /a + +User-agent: gacha-event-tracker +Disallow: /b +`); + expect(isAllowed(robots, UA, "/a")).toBe(false); + expect(isAllowed(robots, UA, "/b")).toBe(false); + expect(isAllowed(robots, UA, "/c")).toBe(true); + }); +}); + +describe("path matching", () => { + test("prefix match, wildcards and the end anchor", () => { + expect(patternMatches("/games/", "/games/Genshin")).toBe(true); + expect(patternMatches("/*.json", "/data/events.json")).toBe(true); + expect(patternMatches("/x$", "/x")).toBe(true); + expect(patternMatches("/x$", "/x/y")).toBe(false); + expect(patternMatches("", "/x")).toBe(false); + }); + + test("longest matching rule wins", () => { + const robots = parseRobots(` +User-agent: * +Disallow: /wiki/ +Allow: /wiki/Event +`); + expect(isAllowed(robots, UA, "/wiki/Special:Random")).toBe(false); + expect(isAllowed(robots, UA, "/wiki/Event")).toBe(true); + }); + + test("allow wins a tie of equal length", () => { + const robots = parseRobots("User-agent: *\nDisallow: /page\nAllow: /page\n"); + expect(isAllowed(robots, UA, "/page")).toBe(true); + }); + + test("the query string is part of the matched target", () => { + const robots = parseRobots("User-agent: *\nDisallow: /*?action=edit\n"); + expect(requestTarget("https://x.test/wiki/Event?action=edit")).toBe( + "/wiki/Event?action=edit", + ); + expect(isAllowed(robots, UA, "/wiki/Event?action=edit")).toBe(false); + expect(isAllowed(robots, UA, "/wiki/Event")).toBe(true); + }); + + test("a path is matched with a leading slash even if given without one", () => { + const robots = parseRobots("User-agent: *\nDisallow: /x\n"); + expect(isAllowed(robots, UA, "x")).toBe(false); + }); +}); + +describe("the sources we actually fetch", () => { + // Game8 opts out of AI-training crawlers by name and leaves everyone else + // alone (CLAUDE.md § Scraping conduct). If that ever changes, this is where + // it should be noticed. + const game8 = parseRobots(` +User-agent: GPTBot +Disallow: / + +User-agent: Google-Extended +Disallow: / + +User-agent: * +Disallow: /admin/ +Disallow: /*?utm_source= +`); + + test("our agent may fetch a Game8 article page", () => { + expect( + isAllowed(game8, UA, "/games/Genshin-Impact/archives/301601"), + ).toBe(true); + }); + + test("the AI-training opt-outs still bind those crawlers", () => { + expect(isAllowed(game8, "GPTBot/1.2", "/games/x")).toBe(false); + expect(isAllowed(game8, "Google-Extended", "/games/x")).toBe(false); + }); + + test("the wildcard rules that do exist are obeyed", () => { + expect(isAllowed(game8, UA, "/admin/")).toBe(false); + expect(isAllowed(game8, UA, "/games/x?utm_source=y")).toBe(false); + }); +}); + +describe("RobotsCache", () => { + function cacheWith( + responder: (url: string) => Response | Promise, + calls: string[] = [], + ) { + return { + calls, + cache: new RobotsCache({ + userAgent: UA, + fetchImpl: async (url) => { + calls.push(url); + return responder(url); + }, + }), + }; + } + + test("fetches robots.txt once per host and reuses it", async () => { + const { cache, calls } = cacheWith( + () => new Response("User-agent: *\nDisallow: /admin\n", { status: 200 }), + ); + expect((await cache.allows("https://game8.co/games/a")).allowed).toBe(true); + expect((await cache.allows("https://game8.co/games/b")).allowed).toBe(true); + expect((await cache.allows("https://game8.co/admin")).allowed).toBe(false); + expect(calls).toEqual(["https://game8.co/robots.txt"]); + expect(cache.fetches).toBe(1); + }); + + test("fetches once per distinct host", async () => { + const { cache, calls } = cacheWith(() => new Response("", { status: 200 })); + await cache.allows("https://game8.co/a"); + await cache.allows("https://endfield.wiki.gg/wiki/Event"); + expect(calls).toEqual([ + "https://game8.co/robots.txt", + "https://endfield.wiki.gg/robots.txt", + ]); + }); + + test("a missing robots.txt means no restrictions", async () => { + const { cache } = cacheWith(() => new Response("nope", { status: 404 })); + const decision = await cache.allows("https://x.test/wiki/Event"); + expect(decision.allowed).toBe(true); + expect(decision.reason).toBe("no robots.txt"); + }); + + test("fails closed on a server error", async () => { + const { cache } = cacheWith(() => new Response("", { status: 503 })); + const decision = await cache.allows("https://x.test/wiki/Event"); + expect(decision.allowed).toBe(false); + expect(decision.reason).toContain("503"); + }); + + test("fails closed when robots.txt is unreachable", async () => { + const { cache } = cacheWith(() => { + throw new Error("ECONNREFUSED"); + }); + const decision = await cache.allows("https://x.test/wiki/Event"); + expect(decision.allowed).toBe(false); + expect(decision.reason).toContain("unreachable"); + }); + + test("expires an entry after its TTL", async () => { + const calls: string[] = []; + let clock = 0; + const cache = new RobotsCache({ + userAgent: UA, + fetchImpl: async (url) => { + calls.push(url); + return new Response("User-agent: *\nDisallow:\n", { status: 200 }); + }, + ttlMs: 1000, + now: () => clock, + }); + + await cache.allows("https://x.test/a"); + clock = 999; + await cache.allows("https://x.test/a"); + expect(calls).toHaveLength(1); + clock = 1001; + await cache.allows("https://x.test/a"); + expect(calls).toHaveLength(2); + }); +}); diff --git a/test/snapshots.test.ts b/test/snapshots.test.ts new file mode 100644 index 0000000..453896f --- /dev/null +++ b/test/snapshots.test.ts @@ -0,0 +1,223 @@ +import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import { mkdtemp, rm, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { + freshnessAt, + hashBody, + SnapshotStore, + type SnapshotMeta, +} from "../src/ingest/snapshots.ts"; + +const SIX_HOURS_MS = 6 * 60 * 60 * 1000; +const T0 = "2026-08-15T00:00:00.000Z"; +const T1 = "2026-08-15T12:00:00.000Z"; + +let root: string; +let store: SnapshotStore; + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), "event-clock-snapshots-")); + store = new SnapshotStore(root); +}); + +afterEach(async () => { + await rm(root, { recursive: true, force: true }); +}); + +function save(html: string, at: string, extra: Partial<{ etag: string | null; lastModified: string | null; eventCount: number | null }> = {}) { + return store.save("genshin-game8-events", { + url: "https://game8.co/games/Genshin-Impact/archives/301601", + html, + etag: extra.etag ?? null, + lastModified: extra.lastModified ?? null, + at, + eventCount: extra.eventCount ?? null, + }); +} + +describe("SnapshotStore", () => { + test("an unfetched source reads as nothing, not as an error", async () => { + expect(await store.read("genshin-game8-events")).toBeNull(); + expect(await store.readMeta("genshin-game8-events")).toBeNull(); + expect(await store.list()).toEqual([]); + }); + + test("stores the body verbatim and its metadata", async () => { + const { changed, meta } = await save("one", T0, { + etag: 'W/"abc"', + lastModified: "Fri, 14 Aug 2026 09:00:00 GMT", + eventCount: 9, + }); + + expect(changed).toBe(true); + expect(meta.contentHash).toBe(hashBody("one")); + expect(meta.eventCount).toBe(9); + + const snapshot = await store.read("genshin-game8-events"); + expect(snapshot?.html).toBe("one"); + expect(snapshot?.meta.etag).toBe('W/"abc"'); + expect(await store.list()).toEqual(["genshin-game8-events"]); + }); + + test("identical bytes are not a change and do not rewrite metadata", async () => { + await save("one", T0, { etag: '"v1"', eventCount: 9 }); + const again = await save("one", T1, { + etag: '"v2"', + eventCount: 9, + }); + + expect(again.changed).toBe(false); + // contentChangedAt still points at the fetch that produced these bytes, + // which is what keeps an unchanged cycle out of the commit log. + expect(again.meta.contentChangedAt).toBe(T0); + expect(again.meta.etag).toBe('"v1"'); + }); + + test("different bytes are a change", async () => { + await save("one", T0, { eventCount: 9 }); + const next = await save("two", T1, { eventCount: 10 }); + + expect(next.changed).toBe(true); + expect(next.meta.contentChangedAt).toBe(T1); + expect((await store.read("genshin-game8-events"))?.html).toBe( + "two", + ); + }); + + test("re-saves when the metadata survived but the body did not", async () => { + await save("one", T0); + await rm(store.bodyPath("genshin-game8-events")); + expect(await store.read("genshin-game8-events")).toBeNull(); + expect((await save("one", T1)).changed).toBe(true); + }); + + test("unreadable metadata is treated as no cache rather than crashing", async () => { + await save("one", T0); + await writeFile(store.metaPath("genshin-game8-events"), "{ truncated"); + expect(await store.readMeta("genshin-game8-events")).toBeNull(); + }); +}); + +describe("conditional requests", () => { + test("emits both validators when both are known", async () => { + const { meta } = await save("one", T0, { + etag: 'W/"abc"', + lastModified: "Fri, 14 Aug 2026 09:00:00 GMT", + }); + expect(store.conditionalHeaders(meta)).toEqual({ + "If-None-Match": 'W/"abc"', + "If-Modified-Since": "Fri, 14 Aug 2026 09:00:00 GMT", + }); + }); + + test("emits only what the server gave us", async () => { + const { meta } = await save("one", T0, { etag: 'W/"abc"' }); + expect(store.conditionalHeaders(meta)).toEqual({ "If-None-Match": 'W/"abc"' }); + }); + + test("a source never fetched sends no validators", () => { + expect(store.conditionalHeaders(null)).toEqual({}); + }); +}); + +describe("the six-hour floor", () => { + const at = (iso: string) => Date.parse(iso); + + test("a source never checked is due", async () => { + const state = await store.readState("genshin-game8-events"); + expect(store.isDue(state, at(T0), SIX_HOURS_MS)).toBe(true); + }); + + test("holds off until six hours have passed", async () => { + await store.recordCheck("genshin-game8-events", { + at: T0, + status: 200, + ok: true, + }); + const state = await store.readState("genshin-game8-events"); + + expect(store.isDue(state, at(T0) + SIX_HOURS_MS - 1, SIX_HOURS_MS)).toBe( + false, + ); + expect(store.isDue(state, at(T0) + SIX_HOURS_MS, SIX_HOURS_MS)).toBe(true); + expect(store.dueAt(state, SIX_HOURS_MS)).toBe(at(T0) + SIX_HOURS_MS); + }); + + test("a failed attempt still counts as an attempt", async () => { + await store.recordCheck("genshin-game8-events", { + at: T0, + status: 503, + ok: false, + }); + const state = await store.readState("genshin-game8-events"); + expect(state.consecutiveFailures).toBe(1); + expect(state.lastConfirmedAt).toBeNull(); + expect(store.isDue(state, at(T0) + 60_000, SIX_HOURS_MS)).toBe(false); + }); + + test("a success clears the failure streak", async () => { + await store.recordCheck("x", { at: T0, status: 503, ok: false }); + await store.recordCheck("x", { at: T1, status: 304, ok: true }); + const state = await store.readState("x"); + expect(state.consecutiveFailures).toBe(0); + expect(state.lastConfirmedAt).toBe(T1); + }); + + test("check bookkeeping lives outside the committed metadata", async () => { + await save("one", T0); + const before = await Bun.file(store.metaPath("genshin-game8-events")).text(); + await store.recordCheck("genshin-game8-events", { + at: T1, + status: 304, + ok: true, + }); + const after = await Bun.file(store.metaPath("genshin-game8-events")).text(); + expect(after).toBe(before); + }); +}); + +describe("freshnessAt", () => { + const meta: SnapshotMeta = { + sourceId: "x", + url: "https://x.test", + contentHash: "abc", + bytes: 3, + etag: null, + lastModified: null, + contentChangedAt: T0, + eventCount: 1, + }; + + test("reports the last confirmation when there is one", () => { + expect( + freshnessAt({ + meta, + state: { + sourceId: "x", + lastCheckedAt: T1, + lastConfirmedAt: T1, + lastStatus: 304, + consecutiveFailures: 0, + }, + html: "", + }), + ).toBe(T1); + }); + + test("never claims to be fresher than the bytes", () => { + expect( + freshnessAt({ + meta, + state: { + sourceId: "x", + lastCheckedAt: null, + lastConfirmedAt: null, + lastStatus: null, + consecutiveFailures: 0, + }, + html: "", + }), + ).toBe(T0); + }); +});