diff --git a/.claude/.gitignore b/.claude/.gitignore new file mode 100644 index 000000000..d80af47c9 --- /dev/null +++ b/.claude/.gitignore @@ -0,0 +1,2 @@ +review-runs/ +__pycache__/ diff --git a/.claude/skills/generate-docs/SKILL.md b/.claude/skills/generate-docs/SKILL.md new file mode 100644 index 000000000..9ffd62411 --- /dev/null +++ b/.claude/skills/generate-docs/SKILL.md @@ -0,0 +1,114 @@ +--- +name: generate-docs +description: Generate or update developer-portal docs from a deployment change (e.g. a helm-charts PR or a new service version) for any service or domain. Discovers what changed and how the services work, verifies the flow live, proposes a change plan, then writes pages and opens a draft docs PR. Use for "document ", "generate docs for the new version", or "/generate-docs". +argument-hint: " --env [--namespace ] [--release ]" +--- + +# Generate docs from a deployment change + +Docs are the spec readers build against, so what gets documented is the **declared intent** +of the change (chart diff, config, the service's own self-description and code), **verified +live**. Where intent and live behaviour disagree, document the intent and report the mismatch +as a deployment finding. Never bake a deployment bug into the docs. + +Nothing here is specific to one domain or protocol. Discover each time; use recipes only as a +head start. + +Helper: `python3 .claude/tools/docrev/docrev.py` (`docrev` below; `docrev -h`). +Environment setup (config in `~/.claude/review-envs/`, `env check`, `env forward`, token +handling) is the same as in the `review-docs` skill, section 2. Follow it. + +## 1. What changed + +- `docrev deploy-diff --pr `: new/modified files, added image/tag/route/ + dependency keys with line numbers. Read the diff itself for config files it lists + (profiles, mappings, service config): those usually carry the intent. +- `docrev inventory --namespace --release `: what is actually running (images, ready + replicas), exposed (routes, admission), and configured (configmaps) for the release. +- Build a list of **changed capabilities**, each tied to evidence: a new service or API + version, new/removed fields, new endpoints or operations, new link types, changed auth, + changed limits. Ignore pure infra changes (resources, replicas, probes) unless they alter + behaviour a client sees. + +## 2. Discover each service + +For every service behind a changed capability: + +1. `recipes/` — if a recipe matches (by image name, protocol, or API style), read it first. +2. Self-description, the way a client would find it (try, don't assume): + OpenAPI/Swagger (`/openapi.json`, `/swagger.json`, `/api-docs`, `/docs`), OGC + `GetCapabilities` / `DescribeRecord` / `DescribeFeatureType` / `GetDomain`, `OPTIONS`, + HAL/JSON:API links, GraphQL introspection, index pages. Use `docrev call` for every request. +3. Declared configuration and code: config files from the chart diff; files inside the + running image via `docrev pod-read` (profile/schema definitions, route tables, mapping + files). `pod-read` masks obvious secrets; still never copy credentials, internal hostnames, + or tokens into docs, recipes, or chat. +4. Existing docs for the same service/domain (`docs/**`): the previous version's pages are the + template and the baseline for "what's new". + +## 3. Intent vs live + +For each capability, compare what the declaration says with what the service does, using +structure rather than values: + +- `docrev shape-diff [--under ]`, where either side can be a + saved response, a file, or a doc block (`doc.md:LINE`). E.g. the previous version's doc + example vs a live response shows what's new/removed; the new profile's declared fields vs a + live record shows whether the deployment actually serves them. +- Try the new capability the way a reader would (filter on a new field, follow a new link + type, call a new operation). A declared field that can't be queried, a link type that never + appears, an operation that errors — each is a **deployment finding**. +- Localise every mismatch hop by hop before naming a cause: public route → proxy → backend. + Query the backend directly (`docrev pod-call`, bypasses routes/proxies/auth) and compare + with the same request through the route; check what each hop actually mounts/loads + (`pod-read`, the Deployment's volumes), not only what the ConfigMaps say. Service logs + often state the cause outright (e.g. an undefined DB column). Treat a restart as a test of + a hypothesis, and re-run `env forward` afterwards (forwards die with their pod). +- Before posting a cause, it must be verified; an unverified hypothesis goes out as a + question, not a finding. +- Walk the intended reader flow end to end (search → metadata → data, or whatever the service + implies), chaining values between steps exactly as `review-docs` section 4 describes, + including its safety rules (writes only with per-request approval; downloads via `--range`). + +## 4. Change plan (stop for approval) + +Present before writing anything: +- Pages to create/update, each with its template page (sibling or previous version) and what + it gets: new sections/steps, table rows with change markers, example requests/responses. +- The reader flow per guide page, as the step list it will have. +- Deployment findings (intent vs live mismatches) with evidence; these are reported, not + documented as behaviour. +- Open questions you can't settle from evidence. Ask; don't pick. + +Wait for the user to approve or adjust. + +## 5. Write + +In a new worktree/branch off the default branch (don't touch the user's working tree): +- Mirror the template page's structure, front matter, tags, admonitions, tabs, and diagram + style. Add new pages to `sidebars.js` next to their siblings. +- Examples: requests use the existing placeholder style (`{SERVICE_URL}`, ``); + responses are captured live, trimmed to what the reader needs, with internal hostnames + replaced by placeholders. Where live output contradicts intent (a deployment finding), + write the example from the intent and leave an HTML comment in the page naming the + finding so reviewers see it. +- Reference tables (fields, enums, parameters) come from the declaration; mark changes vs the + previous version the way existing pages do. +- Keep prose short: step purpose, what to take from the response for the next step, gotchas + found during discovery. + +## 6. Verify and open the PR + +- Run the `review-docs` procedure on the new/changed pages against the same env. Fix doc + findings; keep deployment/env findings for the report. +- `npm run build` must pass with no broken link/anchor warnings for the new pages (the site + config only warns, so read the output). +- Open a **draft** PR (Conventional Commits title, `docs(): ...`). Body: capabilities + documented, reader flows, verification result per page, deployment findings with links to + the deployment PR lines. No pasted code; reference paths. + +## 7. Learn + +If this service kind had no recipe, or the recipe was wrong/incomplete, add or update +`recipes/.md` (format in `recipes/README.md`). Include only what is generic and +verified; no internal hostnames, namespaces, tokens, or product data. diff --git a/.claude/skills/generate-docs/recipes/README.md b/.claude/skills/generate-docs/recipes/README.md new file mode 100644 index 000000000..fabb5dbcd --- /dev/null +++ b/.claude/skills/generate-docs/recipes/README.md @@ -0,0 +1,19 @@ +# Recipes + +Learned notes per service kind, written by `generate-docs` after it documents a service. A +recipe is a head start, never a substitute for discovery: everything in it is re-verified live. + +This repo is public. Recipes hold only generic, verified knowledge: no internal hostnames, +namespaces, tokens, credentials, or product data. + +Format (`.md`, kind = protocol or product, e.g. `ogc-csw-pycsw`): + +```markdown +# +Match: +Self-description: +Declared intent lives in: +Reader flow: +Gotchas: +Docs: +``` diff --git a/.claude/skills/generate-docs/recipes/ogc-csw-pycsw.md b/.claude/skills/generate-docs/recipes/ogc-csw-pycsw.md new file mode 100644 index 000000000..8154f1b4b --- /dev/null +++ b/.claude/skills/generate-docs/recipes/ogc-csw-pycsw.md @@ -0,0 +1,24 @@ +# ogc-csw-pycsw + +Match: image `common/pycsw` or `pycsw`; `POST .../csw` answers `csw:GetRecordsResponse`. + +Self-description: `GetCapabilities`, `DescribeRecord`; queryables per profile via `GetDomain`. + +Declared intent lives in: +- `mappings.py` in the chart (queryable → DB column), +- `pycsw.cfg` (`profiles=`, `table=`, repository `filter=`), +- the profile code inside the image: `/home/pycsw/pycsw/plugins/profiles//` (read with `pod-read`). + +Reader flow: `GetRecords` with a filter (tabs for all / id / type / bbox / polygon / point), paging via `startPosition`/`nextRecord`, then take the record's `mc:links` (by `scheme`) into the next service. + +Gotchas: +- A pycsw with a new profile/mappings that points at the old records table has been observed serving old-shaped records. Compare a live record's shape with the declared profile (`shape-diff --under `), and try filtering on a new field: `Invalid PropertyName` means the new profile isn't really served. +- The declared queryables are listed in GetCapabilities under the `Queryables` constraint (e.g. `MCDEMQueryables`); comparing that list and the `GetRecords` DCP href with the expected version quickly shows whether the route reaches the intended pycsw instance. +- Per-instance nginx in front of pycsw (unified chart) mounts its `location.conf` (`uwsgi_pass `) from a ConfigMap named in `nginx.extraVolumes`; when a second instance doesn't override it, it proxies to the first instance's backend. +- Mappings pointing at columns the table lacks surface as `Invalid query syntax` in CSW and `UndefinedColumn` in the pycsw log; filters on shared columns keep working, which hides the problem. +- A declared queryable isn't necessarily usable everywhere: e.g. `mc:boundingBox` was rejected in spatial filters while `mc:BoundingBox` / `ows:BoundingBox` worked. Test each documented name in the filter types the docs use. +- The profile's bundled XSD (DescribeRecord) can be stale (seen describing another record type); don't take field types from it without checking. +- Polygon filters must be GML3 (`gml:exterior` / `gml:LinearRing` / `gml:posList`); GML2 `outerBoundaryIs`/`coordinates` is rejected with `Missing gml:posList`. +- Errors come back as `ows:ExceptionReport` with HTTP 200. + +Docs: `docs/MapColonies/*/Services/catalog/profile_v*.md`, Step 1 of the DEM/3D guides. diff --git a/.claude/skills/generate-docs/recipes/ogc-wcs-geoserver.md b/.claude/skills/generate-docs/recipes/ogc-wcs-geoserver.md new file mode 100644 index 000000000..cfe3329b8 --- /dev/null +++ b/.claude/skills/generate-docs/recipes/ogc-wcs-geoserver.md @@ -0,0 +1,16 @@ +# ogc-wcs-geoserver + +Match: image `*geoserver*`; `GET .../wcs?request=GetCapabilities` answers `wcs:Capabilities`. + +Self-description: `GetCapabilities` (formats, CRSs, interpolations, `wcs:CoverageId`s), `DescribeCoverage` (envelope, axis labels, grid, nil values). + +Declared intent lives in: chart env (`PROXY_BASE_URL`, extensions), the GeoServer data dir on the PVC (workspace, output limits), the ingestion/publishing service that names coverages. + +Reader flow: GetCapabilities → pick coverage → DescribeCoverage (take `srsName`, `axisLabels`) → GetCoverage (whole / `subset=` per axis / format / optional scaling, `outputCRS`, interpolation). + +Gotchas: +- Coverage ids are prefixed with the workspace (`__`); the prefix is optional in requests. Verify the naming rule the docs claim against real ids. +- Every `xlink:href` in capabilities is built from `PROXY_BASE_URL`; check those hosts resolve to a working route, since clients like QGIS, GDAL and OWSLib follow them. +- Output size limit errors are `ows:ExceptionReport` with HTTP 500; the limit is per-environment config. + +Docs: `docs/ogc/protocols/ogc-wcs.md`, Steps 2–3 of the DEM height extraction guide. diff --git a/.claude/skills/generate-docs/recipes/s3-download-gateway.md b/.claude/skills/generate-docs/recipes/s3-download-gateway.md new file mode 100644 index 000000000..e616025bd --- /dev/null +++ b/.claude/skills/generate-docs/recipes/s3-download-gateway.md @@ -0,0 +1,16 @@ +# s3-download-gateway + +Match: image `common/nginx-s3-gateway`; serves objects of an S3 bucket under a route path. + +Self-description: none; object paths come from catalog links (e.g. `scheme="Download"`). + +Declared intent lives in: chart values (`route.path`, bucket, `authorization.opa`, directory listing flag). + +Reader flow: catalog record → `Download` link → `GET ?token=...`. + +Gotchas: +- Verify with `call --range 1024` rather than downloading whole files. +- Directory paths return 404 when listing is disabled, which is expected. +- Check the no-token response: it should be 401/403, and a 500 is a deployment finding. + +Docs: `docs/MapColonies/DEM/Services/download/README.md`. diff --git a/.claude/skills/review-docs/SKILL.md b/.claude/skills/review-docs/SKILL.md new file mode 100644 index 000000000..7fe156560 --- /dev/null +++ b/.claude/skills/review-docs/SKILL.md @@ -0,0 +1,149 @@ +--- +name: review-docs +description: Review developer-portal docs (a PR or specific files) by running the documented flow or examples against a live environment (e.g. OCP dem-dev) and reporting where the docs, the deployment, or the environment disagree. Use for "review docs PR N against ", "check these docs hold up against the real services", or "/review-docs". +argument-hint: " --env [--deploy-pr ]" +--- + +# Review docs against live services + +The docs are the spec. The goal is to verify that a reader following them, step by step, gets +what the docs promise from the real services, and that the docs are well written. + +Helper: `python3 .claude/tools/docrev/docrev.py` (`docrev` below). It does the +mechanical parts; you do the judgment. Run `docrev -h` for flags. + +## 1. Scope + +- Docs PR: `gh pr view --json files,headRefName` and take changed `docs/**/*.md(x)`. + Read files from the PR head without switching the user's branch: + `git fetch origin pull//head:docrev-pr-` then `git show docrev-pr-:` + into `.claude/review-runs/pr-/`. +- Deployment PR (optional, e.g. helm-charts): `gh pr diff` it. Deployment findings are + anchored to its files/lines. +- Files that are pure reference (profile tables, enum lists) are still in scope: compare them + with what the live service returns in the flows that touch them. + +## 2. Environment + +- Config: `~/.claude/review-envs/.yaml` (outside the repo: it holds internal hostnames and + this repo is public). It maps entry-point placeholders + (`{DEM_CATALOG_SERVICE_URL}`) to URLs, the route that should serve each one, access mode + (`route` or `forward`), and where the token comes from. It never holds the token itself. + ```yaml + name: + namespace: + token: {env: DOCREV_TOKEN_, param: token} # or {file: ~/path}; add header: x-api-key to also send it as a header (header alone: header only) + headers: {x-user-id: } # sent on every request; also fills `` in docs + read_posts: ['/search/', '/route$'] # POST paths the user confirmed have no side effects + read_only: true # e.g. prod: docrev refuses writes even with --allow-write + hosts: [other.example] # our hosts reached only via chained links; `call` sends nothing elsewhere + insecure: true # self-signed dev certs + ca_file: ~/path/chain.pem # instead of insecure, when a server omits its intermediate + placeholders: + DEM_CATALOG_SERVICE_URL: + url: https:/// + route: + access: forward # route | forward + forward: {service: , port: 8080, local_port: 18081, path: /} + aliases: [dem_catalog_url] # other spellings pages use for the same entry point + ``` + Placeholder names match case- and `-`/`_`-insensitively (`` ≡ + `{RASTER_CATALOG_SERVICE_URL}`); add `aliases` only for different names. Add a path to + `read_posts` only after the user confirms it is read-only. +- No config yet: `docrev env discover --namespace [--release ]`, propose a config + from it, get the user's confirmation, then write it. +- Every run: `docrev env check `. Each item is an **env** finding (unadmitted route and who + holds it, service without ready endpoints, missing token). If an entry's route is broken, + use `access: forward` for this run (tell the user) and `docrev env forward `. +- Token missing: ask the user; tell them which env var the config expects. Never write a + token into a file in the repo or echo it into the report. +- Stop forwards at the end: `docrev env stop `. + +## 3. Classify each doc + +`docrev extract --no-content` returns headings (with `step` numbers), tabs, and blocks +(`request`, `example-response`, `diagram`, ...) with line numbers. + +- **Flow**: numbered steps where later steps use output of earlier ones (`kind_hint: flow` is + a hint, confirm by reading). Review the structure as well as the requests: + step numbering is consistent between headings, diagram, and in-text references; every + step's inputs are produced by an earlier step or clearly stated as a prerequisite; the + diagram's edges match what actually feeds what. +- **Examples**: independent snippets (e.g. a protocol page listing requests). Each example + stands alone; a failure affects only that example. +- Mixed pages (e.g. Step 1 with tabs of alternative filters): the tabs are alternatives + within a step; run each, chain from the one the doc says to continue with (or the first + that returns results). + +## 4. Execute + +Run requests in document order with `docrev call --doc --block `. +`extract` recognises curl, bare/multi-line KVP URLs, `POST Request` / `url:` / `body:` blocks, and +XML/JSON bodies whose endpoint is named in the prose just above (`endpoint_from_prose: true`; +check it picked the right one). A `request-body` block has no endpoint nearby: build the curl +yourself and pipe it: `echo "curl ..." | docrev call `. A path-only request (`/route?...`) +needs `--base {VALHALLA_URL}`. + +- **Chaining (flows)**: fill each step's inputs from earlier responses, the way a reader + would: `--sub =`. E.g. `coverageId=srtm30-DTM` → the real + coverage id derived from the catalog record / GetCapabilities; `{WCS_SERVICE_URL}` → the + `WCS_BASE` link of the chosen record. Record where each value came from. If a step's input + can't be obtained from earlier output the way the doc says, that is a finding (the flow is + broken), even if you can still run the step with a value from elsewhere. +- **Illustrative values are not findings.** IDs, product names, coordinates, and dates in the + docs are examples; replace them with real ones and move on. Only the *shape* and + *derivation rule* must hold (e.g. "the id looks like `-`" is a + claim; check it against real ids). +- **Placeholders** like `[COORD1_X]` / `{SRS_IDENTIFIER}`: fill with valid values derived from + earlier responses (e.g. a polygon inside a record's footprint). +- **Environment-specific config is not a doc finding**: size limits, timeouts, hostnames, + counts. Note the observed value; flag only if the doc's *behaviour* claim is wrong (e.g. the + error format or status differs). +- **Safety**: `call` refuses writes (`safety: write`: non-read POST/PUT/PATCH/DELETE). Show + the user the exact request and run with `--allow-write` only after they say yes, per request. + For downloads/large files use `--range 1024` (or `--head`) instead of fetching the file. + To check a "no token needed" claim, re-run with `--no-auth`. + `access: forward` also reroutes a `--sub` to the public URL; add `--no-forward` to test the public route. +- **Success** is not just HTTP 200: an `ows:ExceptionReport` (OGC services often return it + with 200), an empty result where the doc implies results, or a missing link the next step + needs are failures. + +## 5. Compare + +For each step/example, check against the doc: +- Request works as written (after substituting real values). +- Response structure matches the doc's example response: + `docrev shape-diff : [--under ]`. Values may + differ; element/key paths, attributes and link schemes the reader relies on may not. Works + for any XML/JSON API; `--under` aligns a doc snippet with a full response. +- Prose claims: defaults ("default interpolation is linear"), optional/required parameters, + accepted id forms, error messages/status, "save X for step N" actually being needed/usable. +- Reference pages (catalog profiles, enums) vs fields actually returned/queryable. Try + filtering on newly documented fields. +- Writing: typos, wrong API names in client snippets, inconsistent names across pages, + empty table cells, broken internal links/anchors (check the anchor exists). + +Open `saved` response files when the summary isn't enough; don't dump them into the chat. + +## 6. Report + +Each finding has: **kind**, location, one-line statement, evidence (request as run with +`` redacted, status, key response detail). + +| Kind | Meaning | Goes to | +|---|---|---| +| doc | docs wrong/unclear/broken vs the real service | docs PR inline comment | +| deployment | service/chart behaviour contradicts the docs (the spec) | deployment PR (if given), else report | +| env | this environment only (route conflict, pod down, token) | report only | +| unverified | couldn't run (write declined, blocked upstream) | report only | + +Deployment vs doc: the docs are the spec. If the service contradicts a documented behaviour, +it's a deployment finding unless the doc is clearly the one in error (typo, invalid syntax +for the protocol). When genuinely unsure, ask the user instead of choosing. + +Output: +1. Terminal report: verdict per doc (flow holds / breaks at step N / examples x of y pass), + then findings grouped by kind, most severe first. +2. Draft inline comments per PR in `.claude/review-runs//comments--.md`: + `path:line` + a concise comment (no preamble, reference code instead of pasting it). +3. Post only after the user approves, via `gh api` review comments on the PR head commit. diff --git a/.claude/tools/docrev/docrev.py b/.claude/tools/docrev/docrev.py new file mode 100644 index 000000000..1d60b57d1 --- /dev/null +++ b/.claude/tools/docrev/docrev.py @@ -0,0 +1,933 @@ +#!/usr/bin/env python3 +"""Mechanical helpers shared by the review-docs and generate-docs skills. + +Claude does the judgment (discovery, flow vs examples, value chaining, verdicts); +this script only does what must be deterministic and protocol-agnostic: parse +docs, inspect deployments, manage env access, run one request safely, and reduce +any XML/JSON to its structure so two sources can be compared. +""" +import argparse +import base64 +import json +import os +import re +import shlex +import signal +import socket +import ssl +import struct +import subprocess +import sys +import textwrap +import time +import urllib.error +import urllib.parse +import urllib.request +from pathlib import Path + +import yaml + +REPO = Path(__file__).resolve().parents[3] +# Env configs hold internal hostnames and must stay out of this public repo. +ENVS_DIR = Path.home() / ".claude" / "review-envs" +RUNS_DIR = REPO / ".claude" / "review-runs" + +# Docs use several styles: {X_URL}, , , , , [COORD1_X]. +# Lowercase angle forms need a `-`/`_` so plain XML elements (``) don't count. +PLACEHOLDER_RE = re.compile(r"\{[A-Za-z0-9_]+\}|<[A-Z0-9][A-Z0-9 _-]*>|<[a-z][a-z0-9]*[_-][a-z0-9_-]+>||\[[A-Z0-9_]+\]") +URL_START_RE = re.compile(r"^(?:https?://|%s)" % PLACEHOLDER_RE.pattern) +TOKEN_PH_RE = re.compile(r"|\{token\}", re.I) +STEP_RE = re.compile(r"\bstep\s*(\d+(?:\.\d+)?)", re.I) +READ_OPS = re.compile( + r"\b(GetRecords|GetRecordById|DescribeRecord|GetCapabilities|DescribeCoverage|GetCoverage|" + r"DescribeFeatureType|GetFeature|GetMap|GetTile|GetFeatureInfo|GetDomain|GetPropertyValue|" + r"ListStoredQueries|DescribeStoredQueries|GetLegendGraphic)\b", + re.I, +) +MAX_BODY = 5 * 1024 * 1024 + + + +def parse_curl(text): + """Parse a curl command (possibly multi-line) into method/url/headers/body.""" + joined = re.sub(r"\\\r?\n", " ", text.strip()) + try: + argv = shlex.split(joined) + except ValueError as e: + return {"error": f"unparseable curl: {e}"} + if not argv or argv[0] != "curl": + return None + req = {"method": None, "url": None, "headers": {}, "body": None} + it = iter(argv[1:]) + for a in it: + if a in ("-X", "--request"): + req["method"] = next(it, None) + elif a in ("-H", "--header"): + k, _, v = next(it, "").partition(":") + req["headers"][k.strip()] = v.strip() + elif a in ("-d", "--data", "--data-raw", "--data-binary"): + req["body"] = next(it, None) + elif a in ("-o", "--output", "-u", "--user", "-w", "--write-out"): + next(it, None) + elif not a.startswith("-") and req["url"] is None: + req["url"] = a + req["method"] = req["method"] or ("POST" if req["body"] is not None else "GET") + return req + + +def as_url(line): + """`line` as a URL if it is one (quotes/backticks around it allowed), else None.""" + t = line.strip().strip("`'\"") + bare = PLACEHOLDER_RE.sub("X", t) + # `|` and `…` only appear in syntax templates (`osm_ids=[N|W|R],…`), not requests. + if not URL_START_RE.match(t) or t.startswith(("", "[")) or re.search(r"[\s|…]", bare): + return None + return t + + +def bare_url_request(text): + """A URL alone, possibly split one query parameter per line (as OGC KVP pages write it).""" + lines = [l.strip() for l in text.strip().splitlines() if l.strip()] + # A block holding only `` explains a placeholder ("Replace with ..."). + if len(lines) == 1 and PLACEHOLDER_RE.fullmatch(lines[0].strip("`'\"")): + return None + if lines and as_url(lines[0]) and not any(re.search(r"[\s|…]", PLACEHOLDER_RE.sub("X", l)) for l in lines): + return {"method": "GET", "url": "".join(l.strip("`'\"") for l in lines), "headers": {}, "body": None} + return None + + +def body_request(method, url, body): + body = textwrap.dedent(body).strip() or None + headers = {} + if body: + headers["Content-Type"] = "application/json" if body[:1] in "{[" else "application/xml" + # Valhalla-style `.../route?json={}`: the JSON block is the query parameter, not a POST body. + if body and url and re.search(r"=\{\}", url): + compact = json.dumps(json.loads(body), separators=(",", ":")) if body[:1] in "{[" else body + return {"method": "GET", "url": re.sub(r"=\{\}", "=" + urllib.parse.quote(compact), url, count=1), + "headers": {}, "body": None} + return {"method": method, "url": url, "headers": headers, "body": body} + + +def labeled_request(text): + """Blocks written as `POST Request` / `url:` / / `body (XML):` / .""" + lines = text.strip().splitlines() + m = lines and re.match(r"^(GET|POST|PUT|PATCH|DELETE)\s+request\s*:?\s*$", lines[0].strip(), re.I) + if not m: + return None + url, body, i = None, [], 1 + while i < len(lines): + l = lines[i].strip() + if url is None and as_url(l): + parts = [as_url(l)] + while i + 1 < len(lines) and lines[i + 1].strip() and " " not in lines[i + 1].strip() \ + and not lines[i + 1].strip().startswith("<") and re.search(r"[?&]$", parts[-1]): + i += 1 + parts.append(lines[i].strip()) + url = "".join(parts) + elif url is None and not l or re.match(r"^(url|body[^:]*)\s*:?\s*$", l, re.I): + pass + else: + body.append(lines[i]) + i += 1 + try: + return body_request(m.group(1).upper(), url, "\n".join(body)) + except ValueError: + return None + + +INLINE_CODE_RE = re.compile(r"`+([^`]+)`+") + + +def paths_in_code(code): + """URLs, or paths (a `localhost:8002` style host is dropped: the env supplies the base).""" + out = [] + for tok in code.split(): + tok = tok.strip("'\"") + if as_url(tok): + out.append(as_url(tok)) + elif re.match(r"^(?:[\w.-]+:\d+)?/[\w./{}?=&%-]*$", tok): + out.append(tok[tok.index("/"):]) + return out + + +def endpoint_hint(prose): + """The endpoint a body block is sent to, when only the prose above it names it: inline + code (e.g. "make a `POST` request to `/csw`") or a URL alone on a line/block.""" + cands = [] + for line in prose.splitlines(): + if as_url(line): + cands.append(as_url(line)) + for code in INLINE_CODE_RE.findall(line): + cands += paths_in_code(code) + if not cands: + return None + method = next(iter(re.findall(r"\b(GET|POST|PUT|PATCH|DELETE)\b", prose)), "POST") + return method, cands[-1] + + +# Root elements of OGC request bodies; anything else near "request" prose is a sample/fragment. +XML_OPERATION_RE = re.compile(r"^(Get|Describe|Transaction|Lock|List|Harvest)") + + +def xml_root(text): + text = re.sub(r"<\?.*?\?>|", "", text, flags=re.S) + m = re.search(r"<([\w.-]+:)?([\w.-]+)", text) + return m and m.group(2) + + +def labels_response(prose_line): + # A short label ("Response:", "**Example response**"), not a sentence that ends in "response". + return bool(re.fullmatch(r"(?:[\w-]+\s+){0,2}response\s*:?", prose_line.strip().strip("*_:").strip(), re.I)) + + +def extract(md_path): + text = Path(md_path).read_text() + lines = text.splitlines() + headings, blocks, tabs = [], [], [] + current_heading, current_tab, details_depth = None, None, 0 + i = 0 + while i < len(lines): + line = lines[i] + m = re.match(r"^(#{1,6})\s+(.*?)\s*(\{#([\w.-]+)\})?\s*$", line) + if m: + current_heading = {"line": i + 1, "level": len(m.group(1)), "title": m.group(2), + "anchor": m.group(4), "step": None} + s = STEP_RE.search(m.group(2)) + if s: + current_heading["step"] = s.group(1) + headings.append(current_heading) + t = re.search(r'" in line: + current_tab = None + details_depth = max(0, details_depth + len(re.findall(r"]", line)) - line.count("")) + f = re.match(r"^\s*(`{3,})\s*(\w*)(.*)$", line) + if f and f.group(1) in f.group(3): # ```one-line code``` is inline, not a block + f = None + if f: + fence, lang, meta = f.group(1), f.group(2), f.group(3).strip() + start = i + 1 + body = [] + i += 1 + # CommonMark: only a backtick line at least as long as the opener closes it. + while i < len(lines) and not re.match(r"^\s*`{%d,}\s*$" % len(fence), lines[i]): + body.append(lines[i]) + i += 1 + content = "\n".join(body) + title = (re.search(r'title="([^"]*)"', meta) or [None, ""])[1].lower() + prev = next((l for l in reversed(lines[:start - 1]) if l.strip()), "") + is_response = ("response" in title or (details_depth > 0 and "request" not in title) + or labels_response(prev)) + block = { + "line": start, "lang": lang, "meta": meta, "heading": current_heading and current_heading["title"], + "step": current_heading and current_heading["step"], "tab": current_tab, + "role": "example-response" if is_response else lang or "code", + "placeholders": sorted(set(PLACEHOLDER_RE.findall(content))), + "content": content, + } + req = None + if not is_response: + if content.lstrip().startswith("curl"): + req = parse_curl(content) + else: + req = labeled_request(content) or bare_url_request(content) + prose = "\n".join(lines[max(0, start - 7):start - 1]) + is_body = lang == "json" or (lang == "xml" and XML_OPERATION_RE.match(xml_root(content) or "")) + if not req and is_body and re.search(r"request|body|payload", prose, re.I): + hint = endpoint_hint(prose) + try: + req = body_request(*hint, content) if hint else None + except ValueError: + req = None + if req: + block["endpoint_from_prose"] = True + else: + # A body whose endpoint is given elsewhere on the page: Claude builds the request. + block["role"] = "request-body" + if req and not req.get("error") and req.get("url"): + block["role"] = "request" + block["request"] = req + block["safety"] = classify(req) + if lang == "mermaid": + block["role"] = "diagram" + block["diagram_steps"] = sorted(set(STEP_RE.findall(content))) + blocks.append(block) + i += 1 + # A lone endpoint shown before/after "with the following body" is not itself a GET. + body_urls = [b["request"]["url"] for b in blocks if b.get("endpoint_from_prose")] + for b in blocks: + r = b.get("request") + if r and r["method"] == "GET" and "?" not in r["url"] and any(u.startswith(r["url"]) for u in body_urls): + b["role"] = "endpoint" + del b["request"], b["safety"] + step_headings = [h for h in headings if h["step"]] + return { + "file": str(md_path), + "title": next((l.split(":", 1)[1].strip() for l in lines[:15] if l.startswith("title:")), None), + "kind_hint": "flow" if len(step_headings) >= 2 else "examples", + "headings": headings, + "tabs": tabs, + "blocks": blocks, + } + + + +def classify(req, env=None): + method = (req.get("method") or "GET").upper() + url, body = req.get("url") or "", req.get("body") or "" + if method in ("GET", "HEAD", "OPTIONS"): + return "read" + if method == "POST" and (READ_OPS.search(body[:2000]) or READ_OPS.search(url)): + return "read" + # JSON query APIs (routing, geocoding) take reads as POST; the env config lists them + # explicitly because only a human can vouch that a path has no side effects. + path = urllib.parse.urlsplit(url).path + if method == "POST" and any(re.search(p, path) for p in (env or {}).get("read_posts") or []): + return "read" + return "write" + + + +def oc(*args, retries=8): + """oc with retries: some clusters intermittently answer 401 for a valid session.""" + out = None + for _ in range(retries): + out = subprocess.run(["oc", *args], capture_output=True, text=True) + if "Unauthorized" not in out.stderr + out.stdout: + break + time.sleep(1) + return out + + +def load_env(name): + path = ENVS_DIR / f"{name}.yaml" + if not path.exists(): + sys.exit(f"no env config {path}; run `env discover` first") + return yaml.safe_load(path.read_text()) + + +def token_for(env): + src = env.get("token") or {} + if "env" in src: + return os.environ.get(src["env"]) + if "file" in src: + p = Path(os.path.expanduser(src["file"])) + return p.read_text().strip() if p.exists() else None + return None + + +def cmd_env_discover(a): + ns = a.namespace + routes = filter_release(json.loads(oc("get", "routes", "-n", ns, "-o", "json").stdout or "{}").get("items", []), a.release) + out = [] + for r in routes: + conds = (r.get("status", {}).get("ingress") or [{}])[0].get("conditions") or [{}] + out.append({ + "route": r["metadata"]["name"], + "url": f"https://{r['spec']['host']}{r['spec'].get('path', '') or ''}", + "service": r["spec"]["to"]["name"], + "admitted": conds[0].get("status") == "True", + "reason": conds[0].get("reason"), + }) + print(json.dumps(out, indent=2)) + + +def cmd_env_check(a): + env = load_env(a.env) + ns = env["namespace"] + findings = [] + who = oc("whoami") + if who.returncode: + print(json.dumps([{"kind": "env", "issue": "not logged in to cluster", "detail": who.stderr.strip()}])) + return + routes = {r["metadata"]["name"]: r for r in + json.loads(oc("get", "routes", "-n", ns, "-o", "json").stdout or "{}").get("items", [])} + for ph, e in (env.get("placeholders") or {}).items(): + r = routes.get(e.get("route")) if e.get("route") else None + if e.get("route") and not r: + findings.append({"kind": "env", "placeholder": ph, "issue": f"route {e['route']} missing"}) + elif r: + cond = (r.get("status", {}).get("ingress") or [{}])[0].get("conditions") or [{}] + if cond[0].get("status") != "True": + claimers = [n for n, o in routes.items() if n != e["route"] + and o["spec"]["host"] == r["spec"]["host"] and o["spec"].get("path") == r["spec"].get("path")] + findings.append({"kind": "env", "placeholder": ph, + "issue": f"route {e['route']} not admitted ({cond[0].get('reason')})", + "claimed_by": claimers}) + svc = (e.get("forward") or {}).get("service") + if svc: + ep = json.loads(oc("get", "endpoints", svc, "-n", ns, "-o", "json").stdout or "{}") + ready = sum(len(s.get("addresses") or []) for s in ep.get("subsets") or []) + if not ready: + findings.append({"kind": "env", "placeholder": ph, "issue": f"service {svc} has no ready endpoints"}) + if not token_for(env): + findings.append({"kind": "env", "issue": "token not available", "source": env.get("token")}) + print(json.dumps(findings, indent=2)) + + +def port_open(p): + with socket.socket() as s: + return s.connect_ex(("127.0.0.1", p)) == 0 + + +def cmd_env_forward(a): + env = load_env(a.env) + RUNS_DIR.mkdir(parents=True, exist_ok=True) + pidfile = RUNS_DIR / f"{a.env}.forwards.json" + pids = json.loads(pidfile.read_text()) if pidfile.exists() else {} + status = {} + for ph, e in (env.get("placeholders") or {}).items(): + f = e.get("forward") + if not f or (a.only and ph not in a.only): + continue + lp = f["local_port"] + if port_open(lp): + status[ph] = f"already listening on {lp}" + continue + ok = False + for _ in range(10): + p = subprocess.Popen(["oc", "port-forward", "-n", env["namespace"], f"svc/{f['service']}", f"{lp}:{f['port']}"], + stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, start_new_session=True) + for _ in range(10): + if port_open(lp): + ok = True + break + if p.poll() is not None: + break + time.sleep(0.5) + if ok: + pids[ph] = p.pid + break + p.kill() + status[ph] = f"forwarded on {lp}" if ok else "FAILED" + pidfile.write_text(json.dumps(pids)) + print(json.dumps(status, indent=2)) + + +def cmd_env_stop(a): + pidfile = RUNS_DIR / f"{a.env}.forwards.json" + if not pidfile.exists(): + return + for pid in json.loads(pidfile.read_text()).values(): + try: + os.kill(pid, signal.SIGTERM) + except ProcessLookupError: + pass + pidfile.unlink() + + +def norm_ph(p): + return re.sub(r"[-_ ]", "_", p.strip("{}<>[]")).upper() + + +def resolve_url(env, url, forward=True): + """Fill entry-point placeholders and route forwarded prefixes to localhost. + + Placeholder spelling varies between pages (`{RASTER_CATALOG_SERVICE_URL}`, + ``), so names match case- and `-`/`_`-insensitively, + plus any `aliases` the env config lists (e.g. `geocoding_url`).""" + names = {} + for ph, e in (env.get("placeholders") or {}).items(): + for n in [ph, *(e.get("aliases") or [])]: + names[norm_ph(n)] = e["url"].rstrip("/") + url = PLACEHOLDER_RE.sub(lambda m: names.get(norm_ph(m.group(0)), m.group(0)), url) + for e in (env.get("placeholders") or {}).values(): + f = e.get("forward") + if forward and f and e.get("access") == "forward" and url.startswith(e["url"].rstrip("/")): + url = f"http://127.0.0.1:{f['local_port']}{f.get('path', '')}" + url[len(e["url"].rstrip("/")):] + return url + + + +MAGIC = [(b"\x89PNG", "png"), (b"\xff\xd8\xff", "jpeg"), (b"\x1f\x8b", "gzip"), (b"PK\x03\x04", "zip"), + (b"GIF8", "gif"), (b"RIFF", "riff"), (b"%PDF", "pdf"), (b"glTF", "glb"), (b"b3dm", "b3dm")] + + +def summarize(body, ctype): + s = {} + head = body[:4] + if head[:2] in (b"II", b"MM"): + s["tiff"] = tiff_info(body) + return s + kind = next((k for m, k in MAGIC if body.startswith(m)), None) + if kind: + s["binary"] = kind + if kind == "png" and len(body) >= 24: + s["width"], s["height"] = struct.unpack(">II", body[16:24]) + return s + txt = body[:MAX_BODY].decode("utf-8", "replace") + if "xml" in (ctype or "") or txt.lstrip().startswith("]", re.sub(r"<\?.*?\?>|", "", txt, flags=re.S)) or [None, None])[1] + s["exceptions"] = [x.strip() for x in re.findall(r"(?:ExceptionText|ServiceException)[^>]*>\s*([^<]{0,300})", txt) if x.strip()] + s.update({k: v for k, v in re.findall(r'(numberOfRecordsMatched|numberOfRecordsReturned|nextRecord)="(\d+)"', txt)}) + s["links"] = [{"scheme": sc, "name": n, "url": u.strip()} for sc, n, u in + re.findall(r'<\w+:links[^>]*scheme="([^"]*)"[^>]*name="([^"]*)"[^>]*>([^<]*)<', txt)][:20] + s["element_names"] = sorted(set(re.findall(r"<(mc:[A-Za-z0-9]+)[\s>]", txt))) + # What a capabilities document offers (coverages, layers, feature types, tile matrix sets): + # the next step of a flow picks one of these. + ids = {} + for tag, val in re.findall(r"<((?:\w+:)?(?:Identifier|CoverageId|Name|TypeName))>\s*([^<]{1,200}?)\s*" if b[:2] == b"MM" else "<" + off = struct.unpack(e + "I", b[4:8])[0] + n = struct.unpack(e + "H", b[off:off + 2])[0] + tags = {256: "width", 257: "height", 258: "bits", 339: "sample_format"} + out = {} + for k in range(n): + t, ty, c, v = struct.unpack(e + "HHI4s", b[off + 2 + 12 * k: off + 14 + 12 * k]) + if t in tags: + out[tags[t]] = struct.unpack(e + ("H" if ty == 3 else "I"), v[:2] if ty == 3 else v)[0] + if t == 34735: + o = struct.unpack(e + "I", v)[0] + keys = struct.unpack(e + f"{c}H", b[o:o + 2 * c]) + out["epsg"] = [keys[j + 3] for j in range(4, len(keys), 4) if keys[j] in (2048, 3072)] + return out + + +def cmd_call(a): + env = load_env(a.env) + if a.doc: + block = next((b for b in extract(a.doc)["blocks"] if b["line"] == a.block), None) + req = block and block.get("request") + elif a.request: + req = json.loads(Path(a.request).read_text()) + else: + req = parse_curl(sys.stdin.read()) + if not req or req.get("error"): + sys.exit(json.dumps({"error": "could not parse request", "detail": req})) + # Literal substitutions carry values chained from earlier responses (or illustrative + # values swapped for real ones) without editing the doc. + for sub in a.sub or []: + # `OLD=>NEW` when OLD itself contains `=` (e.g. `coverageId=x=>coverageId=y`). + old, _, new = sub.partition("=>") if "=>" in sub else sub.partition("=") + req["url"] = req["url"].replace(old, new) + if req.get("body"): + req["body"] = req["body"].replace(old, new) + safety = classify(req, env) + if safety == "write" and env.get("read_only"): + sys.exit(json.dumps({"error": "env is read_only; writes are never sent", "request": req})) + if safety == "write" and not a.allow_write: + print(json.dumps({"skipped": True, "safety": "write", "request": req})) + return + tok = None if getattr(a, "no_auth", False) else token_for(env) + if req["url"].startswith("/"): + if not a.base: + sys.exit(json.dumps({"error": "relative url; pass --base ", "url": req["url"]})) + req["url"] = a.base.rstrip("/") + req["url"] + url = resolve_url(env, req["url"], forward=not getattr(a, "no_forward", False)) + # The token goes only to the env's own services, never to a third-party example host. + host = (urllib.parse.urlsplit(url).hostname or "").lower() + if host and host not in env_hosts(env): + sys.exit(json.dumps({"error": "host is not in this env; add it to `hosts` if it is ours", "host": host})) + headers = apply_auth(env, tok, req.get("headers") or {}) + tsrc = env.get("token") or {} + if tok: + url = TOKEN_PH_RE.sub(tok, url) + # `header` alone replaces the query param; naming both sends both (prod mixes services). + if (tsrc.get("param") or not tsrc.get("header")) and f"{tsrc.get('param', 'token')}=" not in url: + url += ("&" if "?" in url else "?") + f"{tsrc.get('param', 'token')}={urllib.parse.quote(tok)}" + # `` in a body is usually an XML element (e.g. ``), so only `{X}`/`[X]` count there. + leftover = [p for p in PLACEHOLDER_RE.findall(url) if not TOKEN_PH_RE.fullmatch(p)] + \ + [p for p in PLACEHOLDER_RE.findall(req.get("body") or "") if not p.startswith("<")] + \ + [p for v in headers.values() for p in PLACEHOLDER_RE.findall(v)] + if leftover: + sys.exit(json.dumps({"error": "unfilled placeholders", "placeholders": leftover})) + method = "HEAD" if a.head else req["method"] + if a.range: + headers["Range"] = f"bytes=0-{a.range - 1}" + data = req["body"].encode() if req.get("body") is not None and method != "HEAD" else None + # `ca_file`: a bundle for a server that omits its intermediate cert (verified, unlike `insecure`). + ctx = ssl._create_unverified_context() if env.get("insecure") else \ + ssl.create_default_context(cafile=os.path.expanduser(env["ca_file"])) if env.get("ca_file") else None + started = time.time() + try: + resp = urllib.request.urlopen(urllib.request.Request(url, data=data, headers=headers, method=method), + timeout=a.timeout, context=ctx) + status, rh = resp.status, resp.headers + body = resp.read(MAX_BODY + 1) + except urllib.error.HTTPError as e: + status, rh, body = e.code, e.headers, e.read(MAX_BODY + 1) + except Exception as e: # network-level failure is itself a finding + print(json.dumps({"error": type(e).__name__, "detail": str(e), "url": redact(url, tok)})) + return + RUNS_DIR.mkdir(parents=True, exist_ok=True) + saved = RUNS_DIR / f"resp-{int(started * 1000)}" + # Some services echo the caller's token (e.g. into a `next` link); never keep it on disk. + saved.write_bytes(redact_bytes(body[:MAX_BODY], tok)) + print(json.dumps({ + "url": redact(url, tok), "method": method, "safety": safety, "status": status, + "content_type": rh.get("Content-Type"), "content_length": rh.get("Content-Length"), + "bytes_read": len(body[:MAX_BODY]), "truncated": len(body) > MAX_BODY, + "elapsed_s": round(time.time() - started, 2), "saved": str(saved), + "headers": {k: redact(v, tok) for k, v in rh.items()}, + "summary": summarize(body, rh.get("Content-Type")), + }, indent=2)) + + +def env_hosts(env): + hosts = {"localhost", "127.0.0.1", *(h.lower() for h in env.get("hosts") or [])} + for e in (env.get("placeholders") or {}).values(): + hosts.add((urllib.parse.urlsplit(e["url"]).hostname or "").lower()) + return hosts + + +def apply_auth(env, tok, headers): + """Token header (`token: {header: x-api-key}`) and env-wide `headers`, filling doc + placeholders like `` / `` whose name matches a header.""" + extra = {k.lower(): v for k, v in (env.get("headers") or {}).items()} + theader = (env.get("token") or {}).get("header") + if tok and theader: + extra[theader.lower()] = tok + out = {} + for k, v in headers.items(): + if tok: + v = TOKEN_PH_RE.sub(tok, v) + m = PLACEHOLDER_RE.fullmatch(v.strip()) + key = norm_ph(m.group(0)).lower().replace("_", "-") if m else None + out[k] = extra.get(key) or extra.get(k.lower()) or v if m else v + for k, v in extra.items(): + if k not in {h.lower() for h in out}: + out[k] = v + return out + + +def redact_bytes(b, tok): + if not tok: + return b + for t in (tok, urllib.parse.quote(tok)): + b = b.replace(t.encode(), b"") + return b + + +def redact(s, tok): + return s.replace(tok, "").replace(urllib.parse.quote(tok), "") if tok else s + + + +TAG_RE = re.compile(r"<(/?)([A-Za-z_][\w:.-]*)((?:\s+[^<>]*?)?)(/?)>") +ATTR_RE = re.compile(r"([A-Za-z_][\w:.-]*)\s*=") + + +def xml_shape(text): + """Element/attribute paths of an XML document, values dropped. + + Regex-based on purpose: doc examples are often abbreviated with `...` and + wouldn't parse, and namespace prefixes must be kept as written. + """ + text = re.sub(r"<\?.*?\?>|||]*>", "", text, flags=re.S) + paths, stack = set(), [] + for close, name, attrs, selfclose in TAG_RE.findall(text): + if close: + if name in stack: + while stack and stack.pop() != name: + pass + continue + stack.append(name) + path = "/".join(stack) + paths.add(path) + for a in ATTR_RE.findall(attrs): + if not a.startswith("xmlns") and not a.startswith("xsi:schemaLocation"): + paths.add(f"{path}/@{a}") + if selfclose: + stack.pop() + return paths + + +def json_shape(obj, prefix=""): + paths = set() + if isinstance(obj, dict): + for k, v in obj.items(): + p = f"{prefix}.{k}" if prefix else k + paths.add(p) + paths |= json_shape(v, p) + elif isinstance(obj, list): + for v in obj: + paths |= json_shape(v, prefix + "[]") + return paths + + +def load_source(src): + """A file path, or `doc.md:LINE` for the code block starting at LINE.""" + m = re.match(r"^(.+\.mdx?):(\d+)$", src) + if m: + block = next((b for b in extract(m.group(1))["blocks"] if b["line"] == int(m.group(2))), None) + if not block: + sys.exit(f"no code block at {src}") + return textwrap.dedent(block["content"]) + return Path(src).read_bytes().decode("utf-8", "replace") + + +def shape_of(text): + t = text.lstrip() + if t.startswith("{") or t.startswith("["): + try: + return json_shape(json.loads(t)) + except ValueError: + pass + return xml_shape(text) + + +def cmd_shape(a): + print(json.dumps(sorted(shape_of(load_source(a.source))), indent=1)) + + +def cmd_shape_diff(a): + sa, sb = shape_of(load_source(a.a)), shape_of(load_source(a.b)) + if a.under: + # compare only below the first element with this name, so different wrappers + # (e.g. a doc snippet vs a full response) still line up + def rebase(paths): + out = set() + for p in paths: + parts = p.split("/") + if a.under in parts: + out.add("/".join(parts[parts.index(a.under):])) + return out + sa, sb = rebase(sa), rebase(sb) + if a.ignore: + ig = re.compile(a.ignore) + sa = {p for p in sa if not ig.search(p)} + sb = {p for p in sb if not ig.search(p)} + print(json.dumps({"only_in_a": sorted(sa - sb), "only_in_b": sorted(sb - sa), + "common": len(sa & sb)}, indent=1)) + + +DEPLOY_KEYS = re.compile(r"^\s*-?\s*(repository|tag|image|imageTag|version|host|path|name|url|alias)\s*:\s*(.+?)\s*$") + + +def cmd_deploy_diff(a): + """Summarize a deployment PR/diff: new files, and added image/route/dependency keys.""" + if a.pr: + repo, _, num = a.pr.partition("#") + diff = subprocess.run(["gh", "pr", "diff", num, "-R", repo], capture_output=True, text=True, check=True).stdout + else: + diff = Path(a.diff).read_text() + files, cur, line = {}, None, 0 + for l in diff.splitlines(): + if l.startswith("diff --git"): + cur = l.split(" b/", 1)[1] + files[cur] = {"status": "modified", "added_keys": [], "added": 0, "removed": 0} + elif cur and l.startswith("new file mode"): + files[cur]["status"] = "new" + elif cur and l.startswith("deleted file mode"): + files[cur]["status"] = "deleted" + elif l.startswith("@@"): + line = int(re.search(r"\+(\d+)", l).group(1)) + elif cur and l.startswith("+") and not l.startswith("+++"): + files[cur]["added"] += 1 + m = DEPLOY_KEYS.match(l[1:]) + if m and cur.endswith((".yaml", ".yml")): + files[cur]["added_keys"].append({"line": line, "key": m.group(1), "value": redact_secrets(m.group(2))}) + line += 1 + elif cur and l.startswith("-") and not l.startswith("---"): + files[cur]["removed"] += 1 + elif cur and not l.startswith("\\"): + line += 1 + for f in files.values(): + f["added_keys"] = f["added_keys"][: a.max_keys] + print(json.dumps(files, indent=1)) + + +def release_of(obj): + return (obj["metadata"].get("annotations") or {}).get("meta.helm.sh/release-name") + + +def filter_release(items, release): + """Helm's release annotation is authoritative; the name prefix is only a fallback for + namespaces without Helm-managed objects (a prefix like `dem-` also matches `dem-dev-*`).""" + if not release: + return items + if any(release_of(o) == release for o in items): + return [o for o in items if release_of(o) == release] + return [o for o in items if o["metadata"]["name"].startswith(release + "-")] + + +def cmd_inventory(a): + """Live resources of a namespace (optionally one release): what is actually running and exposed.""" + ns = a.namespace + get = lambda kind: filter_release( + json.loads(oc("get", kind, "-n", ns, "-o", "json").stdout or "{}").get("items", []), a.release) + inv = {"deployments": [], "routes": [], "services": [], "configmaps": []} + for d in get("deployments"): + inv["deployments"].append({ + "name": d["metadata"]["name"], + "images": [c["image"] for c in d["spec"]["template"]["spec"]["containers"]], + "ready": f"{d['status'].get('readyReplicas', 0)}/{d['spec'].get('replicas')}", + }) + for r in get("routes"): + cond = ((r.get("status", {}).get("ingress") or [{}])[0].get("conditions") or [{}])[0] + inv["routes"].append({"name": r["metadata"]["name"], "url": f"https://{r['spec']['host']}{r['spec'].get('path') or ''}", + "service": r["spec"]["to"]["name"], "admitted": cond.get("status") == "True", + "reason": cond.get("reason")}) + for s in get("services"): + inv["services"].append({"name": s["metadata"]["name"], "ports": [p["port"] for p in s["spec"].get("ports", [])]}) + for c in get("configmaps"): + inv["configmaps"].append({"name": c["metadata"]["name"], "keys": sorted((c.get("data") or {}).keys())}) + print(json.dumps(inv, indent=1)) + + +SECRET_RE = re.compile(r"(?i)((?:password|passwd|secret|token|access_?key|secret_?key|api_?key)[\w.-]*\s*[:=]\s*)(\S+)") +URL_CRED_RE = re.compile(r"(\w+://)[^/\s:@]+:[^/\s@]+@") + + +def redact_secrets(text): + return URL_CRED_RE.sub(r"\1***:***@", SECRET_RE.sub(r"\1***", text)) + + +def cmd_pod_read(a): + """Read a file (or list a dir) inside a running workload, with obvious secrets masked.""" + script = ('p="$1"; if [ -d "$p" ]; then ls -la "$p"; ' + 'else head -c %d "$p"; fi' % a.max_bytes) + args = ["exec", "-n", a.namespace, f"deploy/{a.deploy}"] + if a.container: + args += ["-c", a.container] + out = oc(*args, "--", "sh", "-c", script, "sh", a.path) + print(redact_secrets(out.stdout) if out.returncode == 0 else f"error: {out.stderr.strip()}") + + +POD_HTTP = """ +import base64, json, sys, urllib.error, urllib.request as u +r = json.loads(sys.argv[1]) +req = u.Request(r["url"], data=r["body"].encode() if r.get("body") is not None else None, + headers=r.get("headers") or {}, method=r["method"]) +try: + resp = u.urlopen(req, timeout=60); status, ct, body = resp.status, resp.headers.get("Content-Type"), resp.read() +except urllib.error.HTTPError as e: + status, ct, body = e.code, e.headers.get("Content-Type"), e.read() +except (urllib.error.URLError, OSError) as e: + print(json.dumps({"error": str(e)})); sys.exit(0) +print(json.dumps({"status": status, "content_type": ct, "body": base64.b64encode(body[:5242880]).decode()})) +""" + + +def pod_exec_error(stderr): + if re.search(r"python3.*(not found|no such file)", stderr, re.I): + return {"error": "container has no python3; pick another --container or deploy", "detail": stderr.strip()[:300]} + return {"error": "oc exec failed", "detail": stderr.strip()[:300]} + + +def cmd_pod_call(a): + """Send one read request from inside a workload's pod to a local port, bypassing + routes/proxies/auth, to localise which hop of a chain misbehaves.""" + req = parse_curl(sys.stdin.read()) + if not req or req.get("error"): + sys.exit(json.dumps({"error": "could not parse request", "detail": req})) + if classify(req) == "write": + sys.exit(json.dumps({"skipped": True, "safety": "write"})) + parts = urllib.parse.urlsplit(req["url"]) + req["url"] = urllib.parse.urlunsplit(("http", f"127.0.0.1:{a.port}", parts.path or "/", parts.query, "")) + args = ["exec", "-n", a.namespace, f"deploy/{a.deploy}"] + (["-c", a.container] if a.container else []) + out = oc(*args, "--", "python3", "-c", POD_HTTP, json.dumps(req)) + if out.returncode: + sys.exit(json.dumps(pod_exec_error(out.stderr))) + r = json.loads(out.stdout) + if r.get("error"): + sys.exit(json.dumps({"error": "request from pod failed", "url": req["url"], "detail": r["error"]})) + body = base64.b64decode(r["body"]) + RUNS_DIR.mkdir(parents=True, exist_ok=True) + saved = RUNS_DIR / f"pod-resp-{int(time.time() * 1000)}" + saved.write_bytes(body) + print(json.dumps({"url": req["url"], "status": r["status"], "content_type": r["content_type"], + "saved": str(saved), "summary": summarize(body, r["content_type"])}, indent=2)) + + +def main(): + ap = argparse.ArgumentParser(prog="docrev") + sub = ap.add_subparsers(dest="cmd", required=True) + p = sub.add_parser("extract", help="parse a doc into headings/tabs/blocks/requests") + p.add_argument("md") + p.add_argument("--no-content", action="store_true", help="omit block bodies") + p = sub.add_parser("env", help="environment config and access") + esub = p.add_subparsers(dest="ecmd", required=True) + d = esub.add_parser("discover") + d.add_argument("--namespace", required=True) + d.add_argument("--release") + for name in ("check", "forward", "stop"): + e = esub.add_parser(name) + e.add_argument("env") + if name == "forward": + e.add_argument("--only", nargs="*") + p = sub.add_parser("call", help="run one request (curl on stdin or --request JSON)") + p.add_argument("env") + p.add_argument("--request", help="request JSON file (as emitted by extract)") + p.add_argument("--doc", help="doc to take the request from; use with --block") + p.add_argument("--block", type=int, help="line number of the request block in --doc") + p.add_argument("--sub", action="append", metavar="OLD=NEW", help="literal replacement in url and body; OLD=>NEW if OLD contains '='") + p.add_argument("--allow-write", action="store_true") + p.add_argument("--head", action="store_true") + p.add_argument("--range", type=int, help="fetch only the first N bytes") + p.add_argument("--timeout", type=int, default=120) + p.add_argument("--base", help="prefix for a relative request url, e.g. {VALHALLA_URL}") + p.add_argument("--no-forward", action="store_true", help="use the public url even for access: forward entries") + p.add_argument("--no-auth", action="store_true", help="send no token (to check a 'no token needed' claim)") + p = sub.add_parser("shape", help="structure (paths) of an XML/JSON file or doc block (doc.md:LINE)") + p.add_argument("source") + p = sub.add_parser("shape-diff", help="compare structures of two sources") + p.add_argument("a") + p.add_argument("b") + p.add_argument("--under", help="compare only below the first element with this name") + p.add_argument("--ignore", help="regex of paths to ignore") + p = sub.add_parser("deploy-diff", help="summarize a deployment PR or diff") + g = p.add_mutually_exclusive_group(required=True) + g.add_argument("--pr", help="owner/repo#N") + g.add_argument("--diff", help="unified diff file") + p.add_argument("--max-keys", type=int, default=60) + p = sub.add_parser("inventory", help="live deployments/routes/services/configmaps") + p.add_argument("--namespace", required=True) + p.add_argument("--release") + p = sub.add_parser("pod-read", help="read a file or list a dir inside a deployment's pod") + p.add_argument("--namespace", required=True) + p.add_argument("--deploy", required=True) + p.add_argument("--container") + p.add_argument("--max-bytes", type=int, default=200_000) + p.add_argument("path") + p = sub.add_parser("pod-call", help="run one read request (curl on stdin) from inside a pod to a local port") + p.add_argument("--namespace", required=True) + p.add_argument("--deploy", required=True) + p.add_argument("--container") + p.add_argument("--port", type=int, required=True) + a = ap.parse_args() + if a.cmd == "extract": + doc = extract(a.md) + if a.no_content: + for b in doc["blocks"]: + b.pop("content", None) + print(json.dumps(doc, indent=2)) + elif a.cmd == "env": + {"discover": cmd_env_discover, "check": cmd_env_check, + "forward": cmd_env_forward, "stop": cmd_env_stop}[a.ecmd](a) + else: + {"call": cmd_call, "shape": cmd_shape, "shape-diff": cmd_shape_diff, "deploy-diff": cmd_deploy_diff, + "inventory": cmd_inventory, "pod-read": cmd_pod_read, "pod-call": cmd_pod_call}[a.cmd](a) + + +if __name__ == "__main__": + main() diff --git a/.claude/tools/docrev/test_docrev.py b/.claude/tools/docrev/test_docrev.py new file mode 100644 index 000000000..a326c06f5 --- /dev/null +++ b/.claude/tools/docrev/test_docrev.py @@ -0,0 +1,453 @@ +import argparse +import io +import json +import subprocess +import sys +import tempfile +import textwrap +import unittest +import unittest.mock +from contextlib import redirect_stdout +from pathlib import Path + +import docrev + +DOC = textwrap.dedent('''\ + --- + title: Sample Flow + --- + ## Flow diagram + ```mermaid + flowchart LR + a[Step 1] --> b[Step 2] + ``` + + ## Query catalog (Step 1) + + + ```bash + curl --location --request POST '{CATALOG_URL}/csw?token=' \\ + --header 'Content-Type: application/xml' \\ + --data-raw '' + ``` +
+ Response + ```xml + + ``` +
+
+
+ + ## Get coverage (Step 2) {#get-coverage} + ```bash + {WCS_URL}/wcs?request=GetCapabilities&token= + ``` + ''') + + +class ExtractTest(unittest.TestCase): + def setUp(self): + self.path = Path(tempfile.mkdtemp()) / "doc.md" + self.path.write_text(DOC) + self.doc = docrev.extract(self.path) + + def test_flow_detected_from_step_headings(self): + self.assertEqual(self.doc["kind_hint"], "flow") + self.assertEqual([h["step"] for h in self.doc["headings"] if h["step"]], ["1", "2"]) + self.assertEqual(self.doc["headings"][-1]["anchor"], "get-coverage") + + def test_blocks_roles_tabs_and_placeholders(self): + by_line = {b["line"]: b for b in self.doc["blocks"]} + diagram = next(b for b in by_line.values() if b["role"] == "diagram") + self.assertEqual(diagram["diagram_steps"], ["1", "2"]) + curl = next(b for b in by_line.values() if b.get("tab") == "All Records" and b["role"] == "request") + self.assertEqual(curl["request"]["method"], "POST") + self.assertEqual(curl["request"]["headers"]["Content-Type"], "application/xml") + self.assertEqual(curl["safety"], "read") + self.assertIn("{CATALOG_URL}", curl["placeholders"]) + self.assertTrue(any(b["role"] == "example-response" for b in by_line.values())) + bare = [b for b in by_line.values() if b["role"] == "request" and b["request"]["method"] == "GET"] + self.assertEqual(bare[0]["request"]["url"], "{WCS_URL}/wcs?request=GetCapabilities&token=") + + def test_examples_when_no_steps(self): + self.path.write_text("## A\n```bash\ncurl 'http://x/y'\n```\n## B\n") + self.assertEqual(docrev.extract(self.path)["kind_hint"], "examples") + + +class PlaceholderTest(unittest.TestCase): + def test_xml_elements_in_body_are_not_placeholders(self): + env = {"placeholders": {}, "token": {}, "hosts": ["x"]} + req = {"method": "POST", "url": "http://x/csw", "headers": {}, "body": "{SRS}"} + d = Path(tempfile.mkdtemp()) / "r.json" + d.write_text(json.dumps(req)) + out = io.StringIO() + with unittest.mock.patch.object(docrev, "load_env", return_value=env), redirect_stdout(out): + with self.assertRaises(SystemExit) as e: + docrev.cmd_call(argparse.Namespace(env="e", doc=None, block=None, request=str(d), sub=None, + allow_write=False, head=False, range=None, timeout=5)) + self.assertEqual(json.loads(e.exception.code)["placeholders"], ["{SRS}"]) + + def test_token_param_and_header_both_sent(self): + env = {"placeholders": {}, "hosts": ["x"], "token": {"param": "token", "header": "x-api-key"}} + req = {"method": "GET", "url": "http://x/a", "headers": {}, "body": None} + d = Path(tempfile.mkdtemp()) / "r.json" + d.write_text(json.dumps(req)) + sent = {} + + def fake_urlopen(r, **kw): + sent.update(url=r.full_url, headers={k.lower(): v for k, v in r.header_items()}) + raise SystemExit + with unittest.mock.patch.object(docrev, "load_env", return_value=env), \ + unittest.mock.patch.object(docrev, "token_for", return_value="T"), \ + unittest.mock.patch("urllib.request.urlopen", fake_urlopen), redirect_stdout(io.StringIO()): + with self.assertRaises(SystemExit): + docrev.cmd_call(argparse.Namespace(env="e", doc=None, block=None, request=str(d), sub=None, base=None, + allow_write=False, head=False, range=None, timeout=5)) + self.assertEqual((sent["url"], sent["headers"]["x-api-key"]), ("http://x/a?token=T", "T")) + + +class SafetyTest(unittest.TestCase): + def test_classify(self): + self.assertEqual(docrev.classify({"method": "GET", "url": "http://x"}), "read") + self.assertEqual(docrev.classify({"method": "POST", "url": "http://x/csw", "body": ""}), "read") + self.assertEqual(docrev.classify({"method": "POST", "url": "http://x/export", "body": "{}"}), "write") + self.assertEqual(docrev.classify({"method": "DELETE", "url": "http://x/1"}), "write") + + +class ResolveTest(unittest.TestCase): + ENV = {"placeholders": { + "CATALOG_URL": {"url": "https://cat.example/api/v2", "access": "forward", + "forward": {"service": "s", "port": 8080, "local_port": 18081, "path": "/api/v2"}}, + "WCS_URL": {"url": "https://wcs.example", "access": "route"}, + }} + + def test_forwarded_prefix_rewritten_including_chained_urls(self): + self.assertEqual(docrev.resolve_url(self.ENV, "{CATALOG_URL}/csw"), "http://127.0.0.1:18081/api/v2/csw") + self.assertEqual(docrev.resolve_url(self.ENV, "https://cat.example/api/v2/csw?a=1"), + "http://127.0.0.1:18081/api/v2/csw?a=1") + + def test_wcs10_service_exception(self): + xml = b'bad coverage' + self.assertEqual(docrev.summarize(xml, "text/xml")["exceptions"], ["bad coverage"]) + + def test_route_access_left_public(self): + self.assertEqual(docrev.resolve_url(self.ENV, "{WCS_URL}/wcs"), "https://wcs.example/wcs") + + +class SummarizeTest(unittest.TestCase): + def test_ows_exception_and_csw_counts(self): + xml = (b'https://w/wcs') + s = docrev.summarize(xml, "application/xml") + self.assertEqual(s["numberOfRecordsMatched"], "3") + self.assertEqual(s["links"][0]["scheme"], "WCS") + self.assertIn("mc:MCDEMRecord", s["element_names"]) + err = docrev.summarize(b'bad', "") + self.assertEqual(err["exceptions"], ["bad"]) + + +class ShapeTest(unittest.TestCase): + def test_xml_shape_keeps_prefixes_and_tolerates_ellipsis(self): + xml = '1\n...\n' + self.assertEqual(docrev.xml_shape(xml), {"a:Root", "a:Root/a:Item", "a:Root/a:Item/@id", "a:Root/a:Item/a:x"}) + + def test_json_shape_collapses_arrays(self): + self.assertEqual(docrev.json_shape({"a": [{"b": 1}, {"c": 2}]}), {"a", "a[].b", "a[].c"}) + + def test_shape_diff_under_rebases_wrappers(self): + d = Path(tempfile.mkdtemp()) + (d / "doc.xml").write_text("") + (d / "live.xml").write_text("") + out = io.StringIO() + with redirect_stdout(out): + docrev.cmd_shape_diff(argparse.Namespace(a=str(d / "doc.xml"), b=str(d / "live.xml"), under="mc:Rec", ignore=None)) + res = json.loads(out.getvalue()) + self.assertEqual(res["only_in_a"], ["mc:Rec/mc:new"]) + self.assertEqual(res["only_in_b"], ["mc:Rec/mc:old"]) + + +class DeployDiffTest(unittest.TestCase): + DIFF = textwrap.dedent("""\ + diff --git a/c/values.yaml b/c/values.yaml + new file mode 100644 + --- /dev/null + +++ b/c/values.yaml + @@ -0,0 +1,4 @@ + +image: + + repository: common/pycsw + + tag: v7.0.3 + +password: hunter2 + """) + + def test_added_keys_with_line_numbers(self): + d = Path(tempfile.mkdtemp()) / "x.diff" + d.write_text(self.DIFF) + out = io.StringIO() + with redirect_stdout(out): + docrev.cmd_deploy_diff(argparse.Namespace(pr=None, diff=str(d), max_keys=10)) + f = json.loads(out.getvalue())["c/values.yaml"] + self.assertEqual(f["status"], "new") + self.assertEqual([(k["line"], k["key"], k["value"]) for k in f["added_keys"]], + [(2, "repository", "common/pycsw"), (3, "tag", "v7.0.3")]) + + +class ReleaseFilterTest(unittest.TestCase): + def test_annotation_wins_over_prefix(self): + items = [{"metadata": {"name": "dem-a", "annotations": {"meta.helm.sh/release-name": "dem"}}}, + {"metadata": {"name": "dem-dev-b"}}] + self.assertEqual([o["metadata"]["name"] for o in docrev.filter_release(items, "dem")], ["dem-a"]) + self.assertEqual(len(docrev.filter_release(items[1:], "dem")), 1) + + +class RedactTest(unittest.TestCase): + def test_masks_passwords_and_url_credentials(self): + self.assertEqual(docrev.redact_secrets("password: x postgresql://u:p@h/db"), "password: *** postgresql://***:***@h/db") + + +PORTAL_STYLES = textwrap.dedent("""\ + ## Geocode + ```curl + curl -H 'x-api-key: ' '/search/query?q=haifa' + ``` +
+ ```json title="Response" + {"type": "FeatureCollection"} + ``` +
+ + ## WFS + ``` + {WFS_URL}/wfs?service=WFS& + request=GetFeature& + typeNames=a:b + ``` + + Make a `POST` request to `/csw` with this body: + ```xml + + ``` + + ``` + POST Request + url: + {VALHALLA_URL}/route?json={} + body: + {"locations": [1, 2]} + ``` + """) + + +class PortalStylesTest(unittest.TestCase): + def setUp(self): + path = Path(tempfile.mkdtemp()) / "doc.md" + path.write_text(PORTAL_STYLES) + self.reqs = [b for b in docrev.extract(path)["blocks"] if b["role"] == "request"] + self.roles = [b["role"] for b in docrev.extract(path)["blocks"]] + + def test_curl_lang_and_lowercase_placeholders(self): + r = self.reqs[0]["request"] + self.assertTrue(r["url"].startswith("/search")) + self.assertEqual(r["headers"]["x-api-key"], "") + + def test_styled_details_is_response(self): + self.assertIn("example-response", self.roles) + + def test_multiline_kvp_url_joined(self): + self.assertEqual(self.reqs[1]["request"]["url"], "{WFS_URL}/wfs?service=WFS&request=GetFeature&typeNames=a:b") + + def test_body_endpoint_from_prose(self): + b = self.reqs[2] + self.assertTrue(b["endpoint_from_prose"]) + self.assertEqual((b["request"]["method"], b["request"]["url"]), ("POST", "/csw")) + self.assertEqual(b["safety"], "read") + + def test_labeled_block_json_query_param(self): + r = self.reqs[3]["request"] + self.assertEqual(r["method"], "GET") + self.assertIn("route?json=%7B%22locations%22", r["url"]) + + +class AuthAndEnvTest(unittest.TestCase): + ENV = {"placeholders": {"RASTER_CATALOG_SERVICE_URL": {"url": "https://cat/"}, + "GEOCODING_URL": {"url": "https://geo", "aliases": ["geocoding_url"]}}, + "token": {"header": "x-api-key"}, "headers": {"x-user-id": "me"}, + "read_posts": [r"/search/"]} + + def test_placeholder_spelling_variants_resolve(self): + self.assertEqual(docrev.resolve_url(self.ENV, "/csw"), "https://cat/csw") + self.assertEqual(docrev.resolve_url(self.ENV, "/q"), "https://geo/q") + + def test_token_header_and_extra_headers(self): + h = docrev.apply_auth(self.ENV, "T", {"x-api-key": "", "x-user-id": ""}) + self.assertEqual(h, {"x-api-key": "T", "x-user-id": "me"}) + self.assertEqual(docrev.apply_auth(self.ENV, "T", {}), {"x-api-key": "T", "x-user-id": "me"}) + + def test_read_posts_allowlist(self): + req = {"method": "POST", "url": "https://geo/search/query", "body": "{}"} + self.assertEqual(docrev.classify(req), "write") + self.assertEqual(docrev.classify(req, self.ENV), "read") + self.assertEqual(docrev.classify({"method": "POST", "url": "https://x/export-tasks"}, self.ENV), "write") + + + def test_host_allowlist(self): + self.assertEqual(docrev.env_hosts(self.ENV) >= {"cat", "geo", "localhost"}, True) + self.assertNotIn("ows.terrestris.de", docrev.env_hosts(self.ENV)) + + +class ExtractEdgeTest(unittest.TestCase): + def extract(self, text): + path = Path(tempfile.mkdtemp()) / "doc.md" + path.write_text(textwrap.dedent(text)) + return docrev.extract(path)["blocks"] + + def test_syntax_template_is_not_request(self): + blocks = self.extract("""\ + ``` + /lookup?osm_ids=[N|W|R],…,…,& + ``` + """) + self.assertNotEqual(blocks[0]["role"], "request") + + def test_long_fence_and_spaced_lang(self): + blocks = self.extract("""\ + ``````md + ```bash + inner + ``` + `````` + ``` bash + {X_URL}/a?b=c + ``` + """) + self.assertEqual(len(blocks), 2) + self.assertEqual((blocks[1]["lang"], blocks[1]["role"]), ("bash", "request")) + + def test_one_line_fence_is_inline(self): + blocks = self.extract("""\ + ```code``` here + ```bash + {X_URL}/a?b=c + ``` + """) + self.assertEqual([b["role"] for b in blocks], ["request"]) + + def test_placeholder_alone_is_not_request(self): + blocks = self.extract("""\ + ```bash + + ``` + """) + self.assertNotEqual(blocks[0]["role"], "request") + + def test_lone_endpoint_before_body(self): + blocks = self.extract("""\ + We'll invoke a POST GetFeature request + ``` + /wfs + ``` + with the following body: + + ```xml + + ``` + """) + self.assertEqual([b["role"] for b in blocks], ["endpoint", "request"]) + self.assertEqual(blocks[1]["request"]["url"], "/wfs") + + def test_xml_samples_near_request_prose_are_not_bodies(self): + blocks = self.extract("""\ + Find the URL by sending a **GetCapabilities** request. + + ```xml title="Link for WMTS" + '' + ``` + :::warning + To prevent oversized payloads, exceeding the limit triggers: + + ```xml + + + ``` + ::: + We can request a subset of this extent: + ```xml + + ``` + """) + self.assertEqual([b["role"] for b in blocks], ["xml", "xml", "xml"]) + + def test_xml_operation_body_without_endpoint_is_request_body(self): + blocks = self.extract("""\ + Send the request with this body: + ```xml + + + ``` + """) + self.assertEqual(blocks[0]["role"], "request-body") + + def test_plain_response_label(self): + blocks = self.extract("""\ + Response: + + ```xml + + ``` + **Example response:** + ```json + {"a": 1} + ``` + """) + self.assertEqual([b["role"] for b in blocks], ["example-response", "example-response"]) + + def test_sentence_ending_in_response_is_not_label(self): + blocks = self.extract("""\ + We'll add `outputFormat` to each request for a json formatted response + + ``` + {X_URL}/wfs?service=wfs&request=GetCapabilities + ``` + """) + self.assertEqual(blocks[0]["role"], "request") + + +class SummarizeFormatsTest(unittest.TestCase): + def test_png_dimensions(self): + png = b"\x89PNG\r\n\x1a\n" + b"\0\0\0\rIHDR" + (256).to_bytes(4, "big") * 2 + self.assertEqual(docrev.summarize(png, "image/png"), {"binary": "png", "width": 256, "height": 256}) + + def test_geojson(self): + body = json.dumps({"type": "FeatureCollection", "features": [ + {"geometry": {"type": "Point"}, "properties": {"name": "a"}}]}).encode() + s = docrev.summarize(body, "application/json") + self.assertEqual((s["features"], s["geometry_types"], s["property_keys"]), (1, ["Point"], ["name"])) + + def test_capabilities_identifiers(self): + xml = b"orthoa:b" + self.assertEqual(docrev.summarize(xml, "text/xml")["identifiers"], {"ows:Identifier": ["ortho"], "Name": ["a:b"]}) + + + +class PodCallTest(unittest.TestCase): + def test_missing_python_named(self): + err = 'exec failed: unable to start container process: exec: "python3": executable file not found in $PATH' + self.assertIn("python3", docrev.pod_exec_error(err)["error"]) + + def test_other_exec_failures_not_blamed_on_python(self): + err = 'Error from server (NotFound): deployments.apps "nope" not found' + e = docrev.pod_exec_error(err) + self.assertNotIn("python3", e["error"]) + self.assertIn("NotFound", e["detail"]) + + def test_connection_error_reported_by_pod_script(self): + out = subprocess.run([sys.executable, "-c", docrev.POD_HTTP, + json.dumps({"url": "http://127.0.0.1:9/", "method": "GET"})], + capture_output=True, text=True) + self.assertEqual(out.returncode, 0) + self.assertIn("error", json.loads(out.stdout)) + + +if __name__ == "__main__": + unittest.main()