From 9cc6dfc3e430deb10291eddaeb9ef81741ef5bbb Mon Sep 17 00:00:00 2001 From: Just Wicked Code Date: Mon, 9 Mar 2026 20:17:04 +0100 Subject: [PATCH 01/93] Removed git add from husky --- frontend/app/app.vue | 6 ++++++ package.json | 16 ++++++++-------- 2 files changed, 14 insertions(+), 8 deletions(-) create mode 100644 frontend/app/app.vue diff --git a/frontend/app/app.vue b/frontend/app/app.vue new file mode 100644 index 0000000..09f935b --- /dev/null +++ b/frontend/app/app.vue @@ -0,0 +1,6 @@ + diff --git a/package.json b/package.json index d1eacb9..7746821 100644 --- a/package.json +++ b/package.json @@ -1,18 +1,18 @@ { "name": "vroomy", + "private": true, + "scripts": { + "prepare": "husky || true", + "format": "prettier --write .", + "format:check": "prettier --check ." + }, "devDependencies": { "husky": "^9.1.7", "lint-staged": "^16.3.2", "prettier": "^3.8.1" }, "lint-staged": { - "frontend/**/*.{js,jsx,ts,tsx}": "bunx eslint --config frontend/eslint.config.js", - "**/*.{js,jsx,ts,tsx,css,json,md}": "bunx prettier --write" - }, - "private": true, - "scripts": { - "prepare": "husky || true", - "format": "prettier --write .", - "format:check": "prettier --check ." + "frontend/**/*.{js,ts,vue}": "bunx eslint --fix --config frontend/eslint.config.mjs", + "**/*.{js,ts,vue,css,json,md}": "bunx prettier --write" } } From e997cda361a1fa6c751e7801bd68c7f416377b2f Mon Sep 17 00:00:00 2001 From: justwickedcode Date: Tue, 15 Sep 2026 09:18:57 +0200 Subject: [PATCH 02/93] Implemented the scraper for multipla languages and a basic api --- backend/scraper/.env.example | 19 + backend/scraper/Dockerfile | 29 + backend/scraper/README.md | 209 +++- backend/scraper/cmd/crawler/main.go | 65 +- backend/scraper/docker-compose.yml | 2 +- backend/scraper/go.mod | 4 + backend/scraper/handoff.md | 641 ++++++++++-- backend/scraper/internal/crawler/crawler.go | 959 +++++++++++++++++- .../internal/crawler/integration_test.go | 291 ++++++ .../internal/crawler/wikiquote_discovery.go | 405 ++++++++ .../20260524170105_create_url_frontier.sql | 2 +- .../20260913200000_add_quotes_language.sql | 8 + .../20260913230000_add_quotes_word_count.sql | 8 + .../20260915000000_add_quotes_source_url.sql | 5 + backend/scraper/internal/db/redis.go | 22 +- backend/scraper/internal/db/redis_test.go | 50 + backend/scraper/internal/db/store.go | 197 +++- backend/scraper/internal/db/store_test.go | 191 ++++ backend/scraper/internal/dedup/dedup.go | 101 +- backend/scraper/internal/dedup/dedup_test.go | 116 ++- backend/scraper/internal/dedup/language.go | 149 +++ .../scraper/internal/dedup/language_test.go | 96 ++ backend/scraper/internal/fetcher/fetcher.go | 53 +- .../scraper/internal/fetcher/fetcher_test.go | 39 + backend/scraper/internal/models/quote.go | 2 + .../internal/parser/germanwikiquote.go | 208 ++++ .../internal/parser/germanwikiquote_test.go | 42 + backend/scraper/internal/parser/goodreads.go | 104 ++ .../scraper/internal/parser/goodreads_test.go | 111 ++ .../testdata/goodreads_inspirational_tag.html | 146 +++ .../testdata/wikiquote_albert_einstein.html | 35 + backend/scraper/internal/parser/toscrape.go | 32 +- .../scraper/internal/parser/toscrape_test.go | 9 +- backend/scraper/internal/parser/wikilink.go | 72 ++ .../scraper/internal/parser/wikiquote_i18n.go | 194 ++++ .../internal/parser/wikiquote_i18n_test.go | 82 ++ backend/scraper/internal/parser/wikiquotes.go | 116 ++- .../internal/parser/wikiquotes_test.go | 110 ++ backend/scraper/internal/scoring/scoring.go | 38 +- backend/scraper/scripts/backup.sh | 32 + 40 files changed, 4762 insertions(+), 232 deletions(-) create mode 100644 backend/scraper/.env.example create mode 100644 backend/scraper/Dockerfile create mode 100644 backend/scraper/internal/crawler/integration_test.go create mode 100644 backend/scraper/internal/crawler/wikiquote_discovery.go create mode 100644 backend/scraper/internal/db/migrations/20260913200000_add_quotes_language.sql create mode 100644 backend/scraper/internal/db/migrations/20260913230000_add_quotes_word_count.sql create mode 100644 backend/scraper/internal/db/migrations/20260915000000_add_quotes_source_url.sql create mode 100644 backend/scraper/internal/db/redis_test.go create mode 100644 backend/scraper/internal/dedup/language.go create mode 100644 backend/scraper/internal/dedup/language_test.go create mode 100644 backend/scraper/internal/fetcher/fetcher_test.go create mode 100644 backend/scraper/internal/parser/germanwikiquote.go create mode 100644 backend/scraper/internal/parser/germanwikiquote_test.go create mode 100644 backend/scraper/internal/parser/goodreads.go create mode 100644 backend/scraper/internal/parser/goodreads_test.go create mode 100644 backend/scraper/internal/parser/testdata/goodreads_inspirational_tag.html create mode 100644 backend/scraper/internal/parser/testdata/wikiquote_albert_einstein.html create mode 100644 backend/scraper/internal/parser/wikilink.go create mode 100644 backend/scraper/internal/parser/wikiquote_i18n.go create mode 100644 backend/scraper/internal/parser/wikiquote_i18n_test.go create mode 100644 backend/scraper/internal/parser/wikiquotes_test.go create mode 100644 backend/scraper/scripts/backup.sh diff --git a/backend/scraper/.env.example b/backend/scraper/.env.example new file mode 100644 index 0000000..61f2bab --- /dev/null +++ b/backend/scraper/.env.example @@ -0,0 +1,19 @@ +# Copy this file to .env and fill in real values. .env is gitignored — never commit real +# credentials. The placeholder values below (postgres/postgres, redis) are fine for local dev +# against docker-compose.yml; replace them with strong, randomly-generated secrets before any +# real deployment. + +GOOSE_DRIVER=postgres +GOOSE_DBSTRING=postgres://postgres:postgres@localhost:55432/quotes?sslmode=disable +GOOSE_MIGRATION_DIR=./internal/db/migrations + +# Postgres DB +POSTGRES_USER=postgres +POSTGRES_PASSWORD=postgres +POSTGRES_DB=quotes +DATABASE_URL=postgres://postgres:postgres@localhost:55432/quotes?sslmode=disable + +# Redis DB +REDIS_ADDR=localhost:6379 +REDIS_PASSWORD=redis +REDIS_DB=0 diff --git a/backend/scraper/Dockerfile b/backend/scraper/Dockerfile new file mode 100644 index 0000000..695b3fa --- /dev/null +++ b/backend/scraper/Dockerfile @@ -0,0 +1,29 @@ +# Multi-stage build: compile with the full Go toolchain, ship only the static binary in a +# minimal runtime image — the build stage (~1GB+ with the Go toolchain) never ends up in the +# final image, which stays a few MB on top of the base. +FROM golang:1.25-alpine AS build +WORKDIR /src + +# Dependencies cached in their own layer — only re-downloaded when go.mod/go.sum actually change, +# not on every source edit. +COPY go.mod go.sum ./ +RUN go mod download + +COPY . . +# CGO_ENABLED=0 for a fully static binary — runs on the minimal base image below with no libc +# dependency to worry about matching. +RUN CGO_ENABLED=0 GOOS=linux go build -o /crawler ./cmd/crawler + +FROM alpine:3.20 +# ca-certificates: required for the crawler's own outbound HTTPS fetches (Goodreads, Wikiquote) +# to verify TLS certificates — omitting this is a common "works in dev, fails in prod" trap for +# a from-scratch/alpine image, since Go's HTTP client relies on the OS certificate store. +# +# git: the crawler's build-info log line (cmd/crawler/main.go's logBuildInfo) shells out to git +# rev-parse/status. That's meaningless inside a container built from a snapshot with no .git +# directory anyway, so it'll just fail soft and log "could not determine build commit" — fine, +# not worth adding git+the full repo history to the image just to make that one line work. +RUN apk add --no-cache ca-certificates +WORKDIR /app +COPY --from=build /crawler . +ENTRYPOINT ["./crawler"] diff --git a/backend/scraper/README.md b/backend/scraper/README.md index 61207d4..fc83fe6 100644 --- a/backend/scraper/README.md +++ b/backend/scraper/README.md @@ -4,14 +4,16 @@ A scalable web crawler that collects quotes from multiple sources and stores the ## Sources -| Source | Method | Status | Notes | -| ------------------------------------------------------------------------- | ----------------------- | ------------------------- | -------------------------------------------- | -| [quotes.toscrape.com](https://quotes.toscrape.com) | Crawler | ✅ Done | Sandbox site, 100 quotes | -| [Quotable API](https://api.quotable.io) | API Fetcher | ❓ Api at the moment down | 5000+ curated quotes, no auth needed | -| [BrainyQuote](https://brainyquote.com) | Crawler | ⬜ Planned | 100k+ quotes, clean HTML | -| [Wikiquote](https://en.wikiquote.org) | Crawler (MediaWiki API) | ⬜ Planned | Millions of quotes, needs wikitext parser | -| [GoodReads](https://goodreads.com/quotes) | Crawler | ⬜ Planned | Millions of quotes, aggressive bot detection | -| [Kaggle Dataset](https://www.kaggle.com/datasets/akmittal/quotes-dataset) | CSV Import | ⬜ Planned | 500k+ quotes, bulk seed | +This project runs on Goodreads plus fourteen Wikiquote language editions — rather than spreading across every quote site that exists. All are plain Go (`net/http` + goquery), no JS rendering, no bypass of anything. Everything else considered was either cut for cause (see below) or just isn't a priority next to Goodreads' scale. + +| Source | Method | Status | Notes | +| ------------------------------------------------------------------------------------------------------------------------------------ | ---------- | ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| [GoodReads](https://goodreads.com/quotes) | Crawler | ✅ Done | Official sitemap ≈ 5.5M quote URLs. No hard block, but soft-throttles sustained sequential crawling — see note below. Primary volume source. | +| [Wikiquote](https://en.wikiquote.org) | Crawler | ✅ Done | No anti-bot wall (verified live); parses rendered author-page HTML (not wikitext). Self-sustaining primarily via a real MediaWiki title-discovery side channel, supplemented by citation-link `NextURLs` and a random-page fallback — see note below. | +| [Wikiquote (German)](https://de.wikiquote.org) | Crawler | ✅ Done | Same MediaWiki infrastructure as English Wikiquote, reused via the generalized `wikiquoteSite` discovery mechanism (`internal/crawler/wikiquote_discovery.go`), but a structurally different page layout needed its own parser (`internal/parser/germanwikiquote.go`) — see note below. Tags saved quotes `language='de'`. | +| Wikiquote (French / Spanish / Italian / Portuguese / Polish / Swedish / Romanian / Czech / Hungarian / Danish / Norwegian / Finnish) | Crawler | ✅ Done | Same `wikiquoteSite` discovery mechanism, one shared parser (`internal/parser/wikiquote_i18n.go`, `LocalizedWikiquoteParser`) — see "Localized Wikiquote editions" below for why these share one implementation instead of one file each, and its known per-article noise tradeoff. Dutch was checked and deliberately skipped — see that section. | +| [quotes.toscrape.com](https://quotes.toscrape.com) | Test infra | ✅ Done | Not a corpus source — a public scraping sandbox kept solely as the live-fetch integration test target. | +| [Kaggle Dataset](https://www.kaggle.com/datasets/akmittal/quotes-dataset) | CSV Import | ⬜ Planned | Dataset page confirmed live; download needs a Kaggle account/API token (auth, not scraping). 500k+ claimed, unverified until downloaded. | ## Stack @@ -21,6 +23,13 @@ A scalable web crawler that collects quotes from multiple sources and stores the - **HTML Parsing:** goquery - **Queue:** Redis + Asynq (planned) +## Deployment + +`Dockerfile` builds a static binary in a multi-stage build (no CGO, migrations embedded via `//go:embed` — no separate files to copy into the runtime image). For a real production deployment (docker-compose stack alongside `backend/api`, secrets, backups, reverse proxy for the API's HTTPS) see `../../PRODUCTION.md` at the repo root — it documents two real bugs a from-scratch containerized run caught live, worth reading before assuming this "just works" in a container: + +- **`ConnectRedis` used to silently connect to the wrong host** for a plain `host:port` address like a Docker service name (`redis:6379`) — invisible in local dev only because `REDIS_ADDR` there has always literally _been_ `localhost:6379`. Fixed in `internal/db/redis.go`. +- **Real memory need is ~1.25GB**, not a naive guess — the language detector (see "Real language verification" below) loads statistical models into memory, and the actual stable working set was measured live (container memory limit removed, usage observed over a sustained run), not assumed from first principles. + ## Architecture ``` @@ -105,6 +114,18 @@ go test -v ./... # verbose output go test -cover ./... # with coverage ``` +### Live integration tests + +`internal/crawler/integration_test.go` (build tag `integration`) runs against the project's own `docker-compose.yml` Postgres + Redis instead of testcontainers, and includes three tests that do a real HTTP fetch against the live internet (`goodreads.com`, `quotes.toscrape.com`, `en.wikiquote.org`) — no fixtures, no mocking — parsing real pages, saving real quotes, and (for Goodreads and toscrape) pushing a real discovered URL into the Redis `frontier` ZSET. + +```bash +docker compose up -d +go test -tags=integration -count=1 ./internal/crawler/... -v +docker compose down +``` + +State persists in the compose volumes across runs by design (dev env) — the tests are idempotent (dedup on quotes, `ON CONFLICT DO NOTHING` on URLs). + ## URL Frontier The crawler maintains a **URL frontier** — a persistent priority queue of URLs to crawl. PostgreSQL is the source of truth; Redis is the working queue. @@ -134,14 +155,15 @@ CREATE INDEX idx_url_frontier_status_priority ON url_frontier(status, priority); ``` Seed URLs → INSERT into url_frontier (status=pending) ↓ -On startup → load all pending rows → ZADD into Redis Sorted Set (score = priority) +On startup → load all pending rows → ZADD into their per-source Redis Sorted Set ↓ -Fetcher workers → ZPOPMIN from Redis → fetch page → mark status=in_progress in DB +Crawl loop → round-robin across sources (see below) → ZPOPMIN from whichever source's turn + it is → fetch page → mark status=in_progress in DB ↓ -Parser workers → extract quotes + discover next-page URLs +Parser → extract quotes + discover next-page URLs ↓ Quotes → quotes table -New URLs → INSERT INTO url_frontier ON CONFLICT DO NOTHING + ZADD Redis +New URLs → INSERT INTO url_frontier ON CONFLICT DO NOTHING + ZADD into that source's Redis set ↓ Mark URL → status=done or status=failed (increment error_count) ``` @@ -150,19 +172,29 @@ On Redis restart → reload all `status=pending` rows from DB back into Redis (s ### Storage functions -| Function | Status | Notes | -| ---------------------- | ------- | ---------------------------------------------- | -| `db.SaveURL` | ✅ Done | Insert with `ON CONFLICT (url) DO NOTHING` | -| `db.MarkURLDone` | ✅ Done | Sets `status=done`, `last_crawled_at=NOW()` | -| `db.MarkURLFailed` | ✅ Done | Sets `status=failed`, increments `error_count` | -| `db.GetPendingURLs` | ✅ Done | Returns all pending rows ordered by priority | -| `db.WarmFrontierCache` | ✅ Done | Loads pending URLs into Redis on startup | -| `db.PushURL` | ✅ Done | `ZADD frontier ` | -| `db.PopURL` | ✅ Done | `ZPOPMIN frontier` — returns next URL to crawl | +| Function | Status | Notes | +| ---------------------------------------------- | ------- | --------------------------------------------------------------------------------------------------------------------------------------------- | +| `db.SaveURL` | ✅ Done | Insert with `ON CONFLICT (url) DO NOTHING` | +| `db.MarkURLDone` | ✅ Done | Sets `status=done`, `last_crawled_at=NOW()` | +| `db.MarkURLFailed` | ✅ Done | Sets `status=failed`, increments `error_count` | +| `db.GetPendingURLs` | ✅ Done | Returns all pending rows ordered by priority | +| `db.WarmFrontierCache` | ✅ Done | Loads pending URLs into their per-source Redis queue on startup | +| `db.PushURL(ctx, source, url, priority)` | ✅ Done | `ZADD frontier: ` — per-source key, not one shared queue (see below) | +| `db.PopURL(ctx, source)` | ✅ Done | `ZPOPMIN frontier:` — returns the next URL for that specific source | +| `db.HasAnyURLs` | ✅ Done | Any row, any status — used to gate seeding so it only ever runs once (see below) | +| `db.RequeueStuckInProgress` | ✅ Done | Resets `in_progress` → `pending` on startup — recovers work orphaned by a killed/crashed process | +| `db.CountPendingBySource` | ✅ Done | Used to detect when a source has run dry and needs more work discovered | +| `db.GetDiscoveryCursor` / `SetDiscoveryCursor` | ✅ Done | Redis-backed continuation token per source, so title discovery (e.g. Wikiquote's `allpages`) resumes instead of restarting from the beginning | + +### Per-source queues, not one shared priority queue — a real bug, found live + +Frontier state used to be one shared Redis sorted set (`frontier`) across all sources, with `scoring.CalculatePriority` giving each source a different base score so `ZPOPMIN` would naturally prefer the "healthier" source. **This works only as long as no source builds up a real backlog.** Confirmed live, and it's the reason the design changed: once Wikiquote's title discovery (below) gave it hundreds of pending pages, Goodreads' single pending page sat **completely unserved** — 476 Wikiquote pending vs. Goodreads' 1, for as long as the process ran — because Wikiquote's base score was always numerically lower, so `ZPOPMIN` picked a Wikiquote URL first every single time, regardless of how many were queued. + +Fixed by giving each source its own Redis key (`frontier:`) and its own dedicated worker goroutine (`Crawler.runWorker` in `crawler.go`, one per source, launched from `Run()`) instead of one shared loop picking whichever source's turn it was. Priority (`scoring.CalculatePriority`) only orders URLs _within_ a single source's own queue (e.g. by depth) — it has no say in cross-source scheduling at all, which is what actually guarantees every source keeps progressing regardless of how deep another's backlog gets. Verified live after the original round-robin fix: Goodreads advanced two full pages on its own ~20s cadence, cleanly interleaved with Wikiquote fetches every ~6-7s. German Wikiquote (`wikiquote-de`) joined the same setup once added — a third source is just another goroutine launched from `Run()` with its own cooldown and top-up function, nothing else about the mechanism changes. (This started as single-threaded round-robin across a shared loop, then became one goroutine per source once single-threading itself turned out to be the next real bottleneck — see "Concurrent per-source workers" below.) ### Priority Scoring (`internal/scoring`) -Lower score = crawled sooner: +Lower score = crawled sooner **within that source's own queue** — it no longer has any cross-source effect (see above). ``` score = source_base + (depth × DepthPenalty) + (error_count × ErrorPenalty) @@ -176,9 +208,9 @@ score = source_base + (depth × DepthPenalty) + (error_count × ErrorPenalty) | Source | Base Score | | --------------------------- | ---------- | -| `scoring.SourceQuotable` | 1.0 | -| `scoring.SourceBrainyQuote` | 5.0 | -| `scoring.SourceGoodreads` | 20.0 | +| `scoring.SourceWikiquote` | 2.0 | +| `scoring.SourceWikiquoteDE` | 2.0 | +| `scoring.SourceGoodreads` | 10.0 | ### Redis primitives used @@ -189,22 +221,102 @@ score = source_base + (depth × DepthPenalty) + (error_count × ErrorPenalty) > **Note:** Visited URL dedup is handled by the `url_frontier` table itself via `UNIQUE` on `url` + `ON CONFLICT DO NOTHING`. No separate visited set needed. -### Asynq weighted queues (planned) +### Politeness / per-source rate limiting (`crawler.go`) -``` -critical (weight 6) → high-yield, reliable sources (Quotable API) -default (weight 3) → mid-tier sources (BrainyQuote) -low (weight 1) → slow or unreliable sources (Goodreads) -``` +Each source's worker goroutine enforces its own minimum delay between two of its own fetches, **per-source, not one flat number** (`minSourceDelayBySource`) — Goodreads gets 20s, both Wikiquote editions get 6s, because they've earned different levels of caution: a 20-request no-delay burst against Wikiquote (during source research) got a flat `429` after ~11 requests but has otherwise been fine at any pace tested, while Goodreads soft-throttles regardless of pacing. Pooling every source under one delay would either be too lax for Goodreads or needlessly slow for Wikiquote. German Wikiquote reuses English Wikiquote's 6s cooldown outright rather than being separately tuned — same MediaWiki software, no throttling behavior observed yet that would justify treating it differently. + +This is a plain local `lastFetch time.Time` inside `Crawler.runWorker`'s loop, not a shared map — since each source now has its own dedicated goroutine, there's nothing else touching that variable to synchronize against. A source waiting out its cooldown just sleeps; it doesn't block any other source's worker, which is exactly what fixed the "waiting too long between fetches" feeling that motivated this (see "Concurrent per-source workers" below) — previously the entire process was one shared loop, so even with per-source cooldowns and round-robin, a source on cooldown still had to wait for whichever source got served that turn to finish its fetch first. + +**Adaptive backoff — a source backs itself off when it seems rate-limited.** A slow fetch (`slowFetchWarn`, >10s — Goodreads' tarpit signature) or an outright fetch error (including a Wikiquote `429`, which surfaces as `fetcher.Fetch` returning a `*fetcher.StatusError{StatusCode: 429}`) makes `processURL` return `stalled=true`; the calling worker then advances its own `lastFetch` so the next cooldown check yields `stallPenalty` (60s, matching Goodreads' own observed stall duration) instead of the normal per-source delay. Purely local to that source's own goroutine — it no longer needs to "switch to" another source's queue, since that other source was never blocked in the first place. + +**Bounded pacing experiment for Wikiquote (`wikiquoteExperimentalDelay`, currently 4s).** Requested explicitly, with the tradeoff understood: both Wikiquote editions now start at a narrower delay than the proven-safe 6s entry in `minSourceDelayBySource`, chosen from the two data points actually in evidence (6s: proven safe over hundreds of fetches; ~2-3s: confirmed unsafe, the exact pace that produced a live 429 from Wikiquote's discovery API) rather than a third proven-safe number — 4s is a considered guess between them, not a new fact. `fetcher.StatusError` and `fetcher.IsRateLimited(err)` (checking specifically for a 429, not any failure) let `processURL`/`topUpWikiquoteSiteIfEmpty` report a `rateLimited` signal up to `Crawler.runWorker`; the moment either Wikiquote worker actually observes one at this pace, `Crawler.abandonExperimentalPace` permanently widens that worker's `minDelay` back to the safe 6s for the rest of the run — automatically, not something a human needs to be watching logs to catch. Goodreads is unaffected; it never runs at an experimental pace. + +### Concurrent per-source workers + +Originally the crawl loop was one shared loop, round-robining across sources (`Crawler.popNextReady`) so no single source could starve another — a real fix for a real bug (see "Per-source queues" above), but it left a different problem: the loop was still single-threaded, so a slow or stalled fetch on one source (Goodreads' ~60s tarpit is the concrete example) blocked _every_ source's progress until it returned, even a different source's URL that was already off cooldown and ready to go. Round-robin, per-source cooldowns, and adaptive backoff all only act _between_ fetches — none of them can help once a fetch is already in flight. + +Fixed by giving each source in `sourceOrder`'s old place — Goodreads, English Wikiquote, German Wikiquote — its own dedicated goroutine (`Crawler.runWorker`, launched once per source from `Run()`), each running its own independent loop: wait out its own cooldown, pop from its own Redis queue, fetch, parse, save, mark done, repeat. `db.Store`'s underlying `pgxpool.Pool` and `redis.Client` are both already safe for concurrent use by design, so the three goroutines need no additional locking between them — they only ever share the store, and nothing else. Cooldown (`lastFetch`) and, for the two Wikiquote editions, the dead-streak circuit breaker are both plain local variables inside each worker's closure now, not shared maps — since exactly one goroutine ever touches either, there's nothing to synchronize. + +Verified with `go test -race ./...` (clean) plus a full rebuild; not yet measured against a live run for actual throughput gain — see handoff.md. + +**Discovery calls need the same cooldown as page fetches — found live.** `Crawler.runWorker`'s per-source cooldown only updated around an actual page fetch; the `topUp` branch (Wikiquote's MediaWiki API discovery calls) never touched it at all, so consecutive discovery calls (e.g. walking through several already-fully-known categories in a row) were only ~`idlePollInterval` (2s) apart instead of the real per-source delay (6s) — and hit a genuine `429` from Wikiquote's API as a direct result. Fixed by having each top-up function return a `topUpResult{added, calledNetwork, networkFailed}` instead of a bare `bool`: `runWorker` now updates its cooldown after any top-up call that made a real network request (`calledNetwork`), applying the same `stallPenalty` on failure as a page fetch would. Goodreads' top-up never sets `calledNetwork` — it only seeds a tag URL into Postgres/Redis, no request to goodreads.com happens during top-up itself — so it's unaffected. + +### Asynq weighted queues (superseded) + +An Asynq-based weighted-queue design was considered here as the fix for "a stalled fetch blocking other sources," before concurrent per-source goroutines (above) solved the same problem more directly, using only what the codebase already had (no new dependency, no separate worker/queue abstraction layer). Not revisited unless a reason to introduce a real task queue shows up independently (e.g. wanting workers on separate processes/machines, not just separate goroutines in one process). + +### Recursive subcategory discovery + +The curated category list (23 for English, 6 for German) is finite, and confirmed live to actually run out: English Wikiquote's worker went fully idle (0 pending, every curated category's direct members already known) after ~1,572 pages, while Goodreads and German Wikiquote kept progressing fine on their own goroutines — the concurrent-worker fix meant this wasn't a whole-crawler stall, but it was still one source sitting completely idle with real undiscovered content still out there. + +Checked live before building anything: `Category:Writers`, despite being fully mined at the top level, has real, on-topic subcategories the flat list never visited — `Category:Writers by language`, `Category:Gay writers`, `Category:Legal writers`, etc. — and those subcategories have further subcategories of their own (`Writers by language` → `Bengali writers`, `Japanese-language writers`, `Urdu-language writers`, ...). This is a genuinely large additional space, not a marginal one. + +`Crawler.topUpWikiquoteSiteIfEmpty` now walks it: `fetchWikiquoteCategoryMembers` requests `cmtype=page|subcat&cmprop=title|type` — confirmed live that MediaWiki returns a category's own member pages _and_ its direct subcategories together in a single call, tagged by an explicit `type` field, rather than needing two separate requests. Newly found subcategories are queued (breadth-first, so one path doesn't spiral arbitrarily deep before siblings get a turn) and tracked in a visited set scoped to the current top-level branch, guarding against a cyclic or diamond-shaped category graph. Once a category's own pagination (`cmContinue`) is exhausted, the cursor descends into the next queued subcategory; only once a whole branch — the top-level category and everything under it — has nothing left queued does it advance to the next curated category. + +State (`CategoryIndex`, `Current`, `CMContinue`, `Queue`, `Visited`) is packed into the same `wikiquoteDiscoveryCursor` JSON string already persisted per-edition in Redis — no schema change. Each top-up call still makes exactly one HTTP request and is reported back to `runWorker` as one `topUpResult` — deliberately kept 1:1 so the per-source cooldown fix above (each top-up call consumes exactly one cooldown slot) still holds. Combining pages+subcats into one call is a pure efficiency win on top of that, not a relaxation of it: it means a category needs _half_ as many round-trips to fully resolve (one call instead of two), at the same pace as before, rather than resolving at a faster pace. + +Also removed a redundant `idlePollInterval` sleep that used to run _in addition to_ the per-source cooldown after every discovery call — found live to be pure waste (the cooldown wait alone already paces the next attempt correctly), and it was making a chain of already-known categories feel needlessly slow to page through. Deliberately did **not** shorten the per-source discovery cooldown itself below the proven-safe 6s: the live 429 that motivated the original cooldown fix happened at almost exactly a 2-3s pace, so that's the one interval directly confirmed unsafe, not merely untested. + +Verified live (isolated probes against the real API, not assumed): `Category:Writers` → 68 pages + 6 real subcategories in one call; one of those (`Writers by language`, a pure container with 0 direct pages) → 5 further real subcategories (`Bengali writers`, `Japanese-language writers`, etc.) — confirming both that the combined-call approach works end-to-end through the actual Go code, and that the recursion reaches genuine additional content multiple levels deep. + +### Random-page fallback + +Even the recursive category walk can genuinely dry up _for a while_ — confirmed live, not hypothetical: German Wikiquote's queue emptied out (0 pending) while its cursor was deep inside a large `Wissenschaftler` (scientist) subcategory branch, most of whose ~50 sub-subcategories (physicist, chemist, biologist, ...) were already fully known. Every top-level curated category eventually funnels into this same situation once its subtree is mostly mined. + +`Crawler.wikiquoteTopUpFunc` wraps the normal category-walk top-up with a per-edition `emptyStreak` counter (local to the closure set up once per Wikiquote goroutine in `Run()`, not shared state): after `wikiquoteEmptyStreakBeforeRandom` (10) consecutive top-up calls in a row genuinely add nothing new — a network error doesn't count, only a _successful_ call that still found nothing — it falls back to `Crawler.topUpWikiquoteRandomPages`, which calls MediaWiki's `action=query&list=random&rnnamespace=0` for a batch of genuinely random article titles instead. `rnnamespace=0` keeps it to the main article namespace (excludes `Category:`/`Talk:`/`User:`/etc.), but confirmed live it still returns plenty of non-biographical content (topic pages, dates, TV show pages, even the occasional user-babel subpage) since it draws from the _entire_ wiki rather than curated occupation categories — an accepted tradeoff specific to this being a dead-end escape hatch, not the primary mechanism: the existing per-quote filters and the parser itself (no "Quotes" heading → 0 quotes) already handle a low-yield random page harmlessly, and "sometimes fetches a page with nothing on it" beats "sits idle with nothing to try at all." Doesn't touch the category-walk cursor — it's a one-off injection of extra URLs on top, not a replacement; normal category discovery resumes exactly where it left off on the next call. + +Checked whether Goodreads needed the same treatment before building anything for it: live pending counts showed Goodreads sitting at a steady 1 pending (its normal sequential-pagination steady state, not distress) while German Wikiquote was genuinely at 0 — no evidence Goodreads' own tag-cycling fallback is anywhere near its 19-tag limit, so nothing was added there. Goodreads has no MediaWiki-style random-page endpoint to leverage anyway (it's not a wiki); if its tag list ever does prove too small, the natural extension would be more curated tags or the sitemap-based discovery already noted as a future option, not a random-page equivalent. + +### Citation-link discovery (`internal/parser/wikilink.go`) + +Investigated live after a report that discovery within a parsed page "felt broken": on `de.wikiquote.org/wiki/Führer` (a thematic, not biographical, page), the visible on-page links — "Existenz," "Arbeit," "Demokratie" — are topical cross-references embedded in the quote's own body text, not people. Following those blindly would reintroduce exactly the problem an earlier discovery approach at this project already hit once and abandoned (blind `allpages` enumeration pulling in massive non-biographical junk at whole-site scale) — confirmed this is still true today, not just historical. + +The real, precise signal on that same page is different: each quote's _citation_ (e.g. "... - Volker Rühe, Konkret, Heft 2/1998") links the actual person being quoted. `resolveWikiquoteLink` (`internal/parser/wikilink.go`) implements exactly that narrower rule — only links found within a quote's citation are queued, never links found in the quote body itself: + +- Rejects anything not starting with `/wiki/` (an interwiki citation link out to `en.wikipedia.org`, seen live with `class="extiw"`, is already a full absolute URL and gets excluded by this alone). +- Rejects known non-content namespaces (`Category:`/`Kategorie:`, `Special:`/`Spezial:`, `Talk:`/`Diskussion:`, `User:`/`Benutzer:`, `File:`/`Datei:`, `Template:`/`Vorlage:`, `Help:`/`Hilfe:`, `Wikiquote:`, `Portal:`) — both languages' prefixes checked regardless of edition, since a wrong-language namespace name never legitimately collides with a real article title. +- Strips a URL fragment (`/wiki/Aristotle#Politics` → `/wiki/Aristotle`) rather than rejecting it — same page, already worth queueing. + +Wired into both Wikiquote parsers (`wikiquotes.go`, `germanwikiquote.go`), each extracting citation links from their own structurally different citation shape (English: a nested `