From 5bde7ad8861b18eb76a05f64e6822cd34edbd2e7 Mon Sep 17 00:00:00 2001
From: JasonWildMe
Date: Wed, 23 Sep 2026 23:10:37 -0700
Subject: [PATCH 1/7] Add gated submissions API with durable import lifecycle
---
docs/design/2026-09-23-submissions-api.md | 400 ++++
.../2026-09-23-submissions-engineer-brief.md | 124 ++
docs/design/submissions/README.md | 148 ++
docs/design/submissions/examples.json | 292 +++
docs/design/submissions/openapi.yaml | 1588 ++++++++++++++++
docs/design/submissions/pilot-runbook.md | 209 +++
.../submissions/reviews/pr-handoff-review.md | 25 +
.../reviews/stage-1-disposition.md | 38 +
.../submissions/reviews/stage-1-round-1.md | 59 +
.../submissions/reviews/stage-1-round-2.md | 35 +
.../reviews/stage-2-disposition.md | 31 +
.../submissions/reviews/stage-2-round-1.md | 59 +
.../submissions/reviews/stage-2-round-2.md | 29 +
.../reviews/stage-3-disposition.md | 17 +
.../submissions/reviews/stage-3-round-1.md | 63 +
.../submissions/reviews/stage-3-round-2.md | 64 +
.../reviews/stage-4-disposition.md | 18 +
.../submissions/reviews/stage-4-round-1.md | 56 +
.../submissions/reviews/stage-4-round-2.md | 50 +
.../reviews/stage-5-disposition.md | 9 +
.../submissions/reviews/stage-5-round-1.md | 87 +
.../submissions/reviews/stage-5-round-2.md | 55 +
.../submissions/reviews/stage-5-round-3.md | 36 +
.../submissions/reviews/stage-5-round-4.md | 26 +
.../reviews/stage-6-disposition.md | 9 +
.../submissions/reviews/stage-6-round-1.md | 90 +
.../submissions/reviews/stage-6-round-2.md | 66 +
.../submissions/reviews/stage-6-round-3.md | 48 +
.../stage-7-agent-skill-disposition.md | 22 +
.../reviews/stage-7-agent-skill-round-1.md | 53 +
.../reviews/stage-7-agent-skill-round-2.md | 31 +
docs/design/submissions/stage-2-operations.md | 56 +
docs/design/submissions/stage-3-operations.md | 36 +
...26-09-23-submissions-api-implementation.md | 328 ++++
pom.xml | 5 +
scripts/submissions/.gitignore | 1 +
scripts/submissions/check_contract.py | 103 +
scripts/submissions/client.py | 300 +++
scripts/submissions/test_client.py | 178 ++
.../java/org/ecocean/StartupWildbook.java | 5 +
src/main/java/org/ecocean/api/AgentSkill.java | 1 +
src/main/java/org/ecocean/api/AuthToken.java | 25 +-
.../java/org/ecocean/api/Submissions.java | 122 ++
.../java/org/ecocean/api/auth/JwtService.java | 32 +-
.../org/ecocean/api/bulk/BulkImporter.java | 89 +-
.../api/submission/SubmissionException.java | 11 +
.../api/submission/SubmissionFiles.java | 200 ++
.../api/submission/SubmissionImporter.java | 67 +
.../api/submission/SubmissionJobs.java | 277 +++
.../api/submission/SubmissionJson.java | 120 ++
.../api/submission/SubmissionPolicy.java | 31 +
.../api/submission/SubmissionResources.java | 16 +
.../api/submission/SubmissionStore.java | 215 +++
.../api/submission/SubmissionValidator.java | 87 +
.../api/submission/SubmissionWorker.java | 70 +
.../SubmissionAuthenticationFilter.java | 100 +
.../WildbookTokenAuthenticationFilter.java | 4 +
.../org/ecocean/submission/Submission.java | 97 +
.../resources/agent-skills/api-reference.md | 4 +
src/main/resources/agent-skills/index.md | 14 +-
.../agent-skills/submit-sightings.md | 419 +++++
src/main/resources/openapi.yaml | 1651 +++++++++++++++++
.../org/ecocean/submission/package.jdo | 36 +
src/main/webapp/WEB-INF/web.xml | 13 +
.../ecocean/api/AgentSkillContentTest.java | 21 +-
.../api/AuthTokenSubmissionScopeTest.java | 51 +
.../java/org/ecocean/api/SubmissionsTest.java | 63 +
.../org/ecocean/api/bulk/BulkApiPostTest.java | 69 +
.../BulkImporterSubmissionBoundaryTest.java | 43 +
.../bulk/BulkSubmissionCompatibilityTest.java | 80 +
.../api/submission/SubmissionFilesTest.java | 81 +
.../api/submission/SubmissionJsonTest.java | 36 +
.../api/submission/SubmissionPolicyTest.java | 22 +
.../api/submission/SubmissionStoreDbTest.java | 410 ++++
.../submission/SubmissionValidatorTest.java | 45 +
.../SubmissionAuthenticationFilterTest.java | 98 +
76 files changed, 9525 insertions(+), 44 deletions(-)
create mode 100644 docs/design/2026-09-23-submissions-api.md
create mode 100644 docs/design/2026-09-23-submissions-engineer-brief.md
create mode 100644 docs/design/submissions/README.md
create mode 100644 docs/design/submissions/examples.json
create mode 100644 docs/design/submissions/openapi.yaml
create mode 100644 docs/design/submissions/pilot-runbook.md
create mode 100644 docs/design/submissions/reviews/pr-handoff-review.md
create mode 100644 docs/design/submissions/reviews/stage-1-disposition.md
create mode 100644 docs/design/submissions/reviews/stage-1-round-1.md
create mode 100644 docs/design/submissions/reviews/stage-1-round-2.md
create mode 100644 docs/design/submissions/reviews/stage-2-disposition.md
create mode 100644 docs/design/submissions/reviews/stage-2-round-1.md
create mode 100644 docs/design/submissions/reviews/stage-2-round-2.md
create mode 100644 docs/design/submissions/reviews/stage-3-disposition.md
create mode 100644 docs/design/submissions/reviews/stage-3-round-1.md
create mode 100644 docs/design/submissions/reviews/stage-3-round-2.md
create mode 100644 docs/design/submissions/reviews/stage-4-disposition.md
create mode 100644 docs/design/submissions/reviews/stage-4-round-1.md
create mode 100644 docs/design/submissions/reviews/stage-4-round-2.md
create mode 100644 docs/design/submissions/reviews/stage-5-disposition.md
create mode 100644 docs/design/submissions/reviews/stage-5-round-1.md
create mode 100644 docs/design/submissions/reviews/stage-5-round-2.md
create mode 100644 docs/design/submissions/reviews/stage-5-round-3.md
create mode 100644 docs/design/submissions/reviews/stage-5-round-4.md
create mode 100644 docs/design/submissions/reviews/stage-6-disposition.md
create mode 100644 docs/design/submissions/reviews/stage-6-round-1.md
create mode 100644 docs/design/submissions/reviews/stage-6-round-2.md
create mode 100644 docs/design/submissions/reviews/stage-6-round-3.md
create mode 100644 docs/design/submissions/reviews/stage-7-agent-skill-disposition.md
create mode 100644 docs/design/submissions/reviews/stage-7-agent-skill-round-1.md
create mode 100644 docs/design/submissions/reviews/stage-7-agent-skill-round-2.md
create mode 100644 docs/design/submissions/stage-2-operations.md
create mode 100644 docs/design/submissions/stage-3-operations.md
create mode 100644 docs/plans/2026-09-23-submissions-api-implementation.md
create mode 100644 scripts/submissions/.gitignore
create mode 100644 scripts/submissions/check_contract.py
create mode 100644 scripts/submissions/client.py
create mode 100644 scripts/submissions/test_client.py
create mode 100644 src/main/java/org/ecocean/api/Submissions.java
create mode 100644 src/main/java/org/ecocean/api/submission/SubmissionException.java
create mode 100644 src/main/java/org/ecocean/api/submission/SubmissionFiles.java
create mode 100644 src/main/java/org/ecocean/api/submission/SubmissionImporter.java
create mode 100644 src/main/java/org/ecocean/api/submission/SubmissionJobs.java
create mode 100644 src/main/java/org/ecocean/api/submission/SubmissionJson.java
create mode 100644 src/main/java/org/ecocean/api/submission/SubmissionPolicy.java
create mode 100644 src/main/java/org/ecocean/api/submission/SubmissionResources.java
create mode 100644 src/main/java/org/ecocean/api/submission/SubmissionStore.java
create mode 100644 src/main/java/org/ecocean/api/submission/SubmissionValidator.java
create mode 100644 src/main/java/org/ecocean/api/submission/SubmissionWorker.java
create mode 100644 src/main/java/org/ecocean/security/SubmissionAuthenticationFilter.java
create mode 100644 src/main/java/org/ecocean/submission/Submission.java
create mode 100644 src/main/resources/agent-skills/submit-sightings.md
create mode 100644 src/main/resources/org/ecocean/submission/package.jdo
create mode 100644 src/test/java/org/ecocean/api/AuthTokenSubmissionScopeTest.java
create mode 100644 src/test/java/org/ecocean/api/SubmissionsTest.java
create mode 100644 src/test/java/org/ecocean/api/bulk/BulkImporterSubmissionBoundaryTest.java
create mode 100644 src/test/java/org/ecocean/api/bulk/BulkSubmissionCompatibilityTest.java
create mode 100644 src/test/java/org/ecocean/api/submission/SubmissionFilesTest.java
create mode 100644 src/test/java/org/ecocean/api/submission/SubmissionJsonTest.java
create mode 100644 src/test/java/org/ecocean/api/submission/SubmissionPolicyTest.java
create mode 100644 src/test/java/org/ecocean/api/submission/SubmissionStoreDbTest.java
create mode 100644 src/test/java/org/ecocean/api/submission/SubmissionValidatorTest.java
create mode 100644 src/test/java/org/ecocean/security/SubmissionAuthenticationFilterTest.java
diff --git a/docs/design/2026-09-23-submissions-api.md b/docs/design/2026-09-23-submissions-api.md
new file mode 100644
index 0000000000..419f55b427
--- /dev/null
+++ b/docs/design/2026-09-23-submissions-api.md
@@ -0,0 +1,400 @@
+# Wildbook submissions API
+
+Status: senior engineer accepted the sibling API, new-API defaults and limited
+partner pilot on 2026-09-23, as relayed by the user. Detailed implementation
+mechanics remain proposed; no runtime changes implemented.
+
+Reviewed 2026-09-23 against checkout `24cc99aede`, the Wildbook development skill,
+the import-format skill, and the [lead engineer's proposal](https://qa.wildme.org/bulk-import-next.html).
+The proposal was retrieved directly; deployment behavior was not tested.
+
+## Recommendation
+
+Add a small `/api/v3/submissions` API for authenticated data intake. Reuse the
+bulk-import row contract and Java importer. Keep the existing React bulk-import
+routes, defaults, authentication, and response shapes stable.
+
+A submission is a server-owned draft containing metadata and uploaded media.
+Clients create a draft, upload files, validate it, commit it once, and retrieve
+the resulting Wildbook records. One encounter and 200 encounters use the same
+workflow. A submission is an intake envelope, not a new biological entity.
+
+This adds orchestration around the existing importer, not another implementation
+of encounter creation. The main new responsibilities are ownership before an
+ImportTask exists, immutable commit input, retry protection, and a predictable
+machine-readable lifecycle.
+
+Initial scope: trusted integrations, agents acting for authenticated users, and
+humans using a CLI or future form. Anonymous public submission, general record
+updates, automatic identity decisions, and replacement of the working bulk UI
+are separate projects. Existing import-supported individual/occurrence linking
+remains available only within the caller's permissions.
+
+## What already exists
+
+| Capability | Evidence in this checkout | Design consequence |
+| --- | --- | --- |
+| JSON intake | `api/BulkImport.java#doPost`: object rows, or `fieldNames` plus array rows | No CSV/XLSX parser needed for programmatic intake |
+| Field and row validation | `api/bulk/BulkValidator.java`, `BulkImportUtil.validateRow` | Reuse these as the domain-validation authority |
+| Validation mode | `validateOnly` with field/row diagnostics | Reuse validation logic; new preview contract must also describe media checks |
+| Creation | `BulkImporter.createImport`, `UploadedFiles.makeMediaAsset` | Reuse encounter, occurrence, individual, project, and media handling |
+| Async tracking | `ImportTask`; `processInBackground`; status GET | Link submissions to ImportTasks; distinguish data import from IA completion |
+| Uploads | `UploadServlet`, `UploadPaths`, `FlowInfoStorage` | Preserve working browser upload path; reuse mechanics where appropriate |
+| JWT verification | `WildbookTokenAuthenticationFilter`, `JwtService`, `AuthToken` | Reuse verification and identity resolution, with explicit write authorization |
+| Import visibility | `ServletUtilities.isUserAuthorizedForImportTask` | Existing policy includes creator, collaboration, and organization/admin checks |
+
+Paths in this table are relative to `src/main/java/org/ecocean/`.
+
+The current servlet contains orchestration as well as validation. Its helper
+methods are not already a clean service API. `BulkImporter.createImport` also
+starts media-child/indexing work, while its caller commits the main transaction.
+Any new execution adapter must preserve and test those lifecycle dependencies.
+Calling `createImport` alone is not the complete import workflow.
+
+## How this develops the engineer's proposal
+
+Keep its core decisions: additive backend work, normalized JSON rows, reuse of
+the importer, simple multipart uploads, documented resumable uploads, and no
+initial React changes. Make the following refinements:
+
+1. **Separate the new orchestration contract.** A sibling submissions endpoint
+ avoids adding new draft/commit semantics to a working servlet. The cost is a
+ small new resource and adapter; the benefit is that new defaults and status
+ codes do not affect browser imports. Merely documenting and token-enabling
+ the old POST is smaller, but does not provide the retry and draft lifecycle
+ expected by unattended clients.
+2. **Write permission must be explicit.** A different filter name or URL does
+ not distinguish a read token from an import token. Require a signed import
+ capability issued after explicit authentication and permission checks.
+ Existing tokens without that capability remain ineligible for submissions.
+3. **Keep location policy local to the new contract initially.** Require a
+ configured `Encounter.locationID` for the new MVP. Do not change the shared
+ required-field set merely to make the new API stricter. Shared behavior
+ changes need their own compatibility review.
+4. **Count media per encounter after grouping rows.** A submission-wide file
+ count is not `maximumMediaCountEncounter`. Apply distinct limits for file
+ bytes, request bytes, draft storage, row count, and encounter media count.
+5. **Define retry and filename semantics.** An existing filename or ImportTask
+ ID is not a sufficient idempotency contract. Verify content, freeze the draft,
+ and track the commit operation durably.
+
+Some proposal baseline details differ from this checkout: upload requests already
+use configured chunk/request byte bounds; `FlowInfoStorage` keys include both
+identifier and staging path; task authorization is broader than creator-only.
+Token TTL is selected by server configuration, not by a client request parameter.
+Chunk/request bounds still need to be distinguished from complete-file and
+submission quotas in the new contract. The skill's 200-row recommendation is a
+useful starting default, not evidence of a universal backend maximum.
+
+## API contract
+
+All routes below are **proposed**, not available today. Publish them through the
+existing OpenAPI documentation mechanism. Require authentication and return JSON
+errors, including for expired sessions; never redirect an API client to login.
+
+| Method and route | Purpose and result |
+| --- | --- |
+| `GET /api/v3/submissions/capabilities` | Contract version, enabled operations, row schema, configured values and limits |
+| `POST /api/v3/submissions` | Create owned draft; `201`, ID, revision, expiry, links; require `Idempotency-Key` |
+| `GET /api/v3/submissions/{id}` | Draft/operation status and links; present immediately after creation |
+| `PUT /api/v3/submissions/{id}/rows` | Replace complete normalized row set in a draft; require `If-Match` revision |
+| `POST /api/v3/submissions/{id}/files` | Stream one multipart file; require draft revision and return new revision |
+| `GET /api/v3/submissions/{id}/files` | Manifest: name, size, digest, upload/validation state |
+| `POST /api/v3/submissions/{id}/validate` | Validate a revision; `200` with `valid`, issues, counts, validation ID |
+| `POST /api/v3/submissions/{id}/commit` | Commit validated revision; require `If-Match` and `Idempotency-Key`; `202`, status URL |
+| `GET /api/v3/submissions/{id}/results` | Paginated record IDs, row references, diagnostics and downstream status |
+| `DELETE /api/v3/submissions/{id}` | Cancel an uncommitted draft and expire staged files; never delete imported records |
+
+Mutation during validation/commit is serialized per submission. A stale revision
+returns `412`; conflicting operation or filename content returns `409`.
+Malformed JSON returns `400`, unacceptable commit data `422`, oversized input
+`413`, and quota/concurrency throttling `429` with `Retry-After`. Validation
+returning HTTP 200 means the check ran, not that the data passed.
+
+Example draft creation body:
+
+```json
+{
+ "contractVersion": "1",
+ "source": {"name": "field-survey-tool", "batchId": "survey-2026-09-23-a"},
+ "processing": {"mode": "import-only"}
+}
+```
+
+Example body for `PUT .../{id}/rows`:
+
+```json
+{
+ "rows": [
+ {
+ "clientRowId": "observation-001",
+ "fields": {
+ "Encounter.year": 2026,
+ "Encounter.month": 9,
+ "Encounter.day": 23,
+ "Encounter.locationID": "configured-location-id",
+ "Encounter.genus": "Loxodonta",
+ "Encounter.specificEpithet": "africana",
+ "Encounter.mediaAsset0": "observation-001.jpg"
+ }
+ }
+ ]
+}
+```
+
+`fields` passes to the current object-row validator. The small wrapper carries
+provenance without inventing new bulk field names. `clientRowId` is unique within
+a submission and survives validation and result mapping. The location and
+taxonomy in this example must be replaced with values configured on the target
+installation. Default submitter is the authenticated user; declaring another
+owner requires explicit authority. Source metadata never establishes authority.
+
+Expose three processing choices: `import-only`, `detect`, and
+`detect-and-identify`, mapped to existing skip flags. Default the new API to
+`import-only` to avoid unrequested processing; retain legacy defaults. Discovery
+lists which choices are supported by the installation/taxon. A match result is
+not automatic confirmation of an individual's identity.
+
+The MVP supports image-backed encounters. Require at least one completed image
+reference per resulting encounter. Metadata-only records, videos, and externally
+hosted URL ingestion can be added as explicit capabilities later. Preserve date
+precision: a year alone remains a year, not an invented January 1 observation.
+
+## Validation and discovery
+
+Run three layers, returning stable issue codes plus readable messages:
+
+1. **Envelope and permission checks:** nonempty bounded rows, unique client row
+ IDs, ownership, configured location, field allowlist, authorized links to
+ existing records/projects, and complete manifest references.
+2. **Existing domain checks:** `BulkImportUtil.validateRow` and `BulkValidator`.
+ Copy input before validation because validation can mutate JSON key sets.
+ Resolve defaults once and show the effective values in the preview.
+3. **Media and aggregate checks:** decode staged images without creating domain
+ objects; validate actual bytes and digest, merged encounter media counts,
+ and total job limits. Do not describe filename existence as image validity.
+
+Unknown field names are errors in the new API. Reject invalid rows as a whole
+submission for the MVP; do not expose permissive legacy tolerance knobs yet.
+Validation can write its report and operational audit, but creates no Encounters,
+MediaAssets, Individuals, Projects, or IA jobs.
+
+Discovery should derive field names, synonyms, types and enums from existing
+validator/configuration sources where available. Add a thin metadata description
+where those sources lack types/help text, with drift checks; do not hand-maintain
+a second independent validator. Expose conditional requirements and dynamic
+measurements/keywords as well as simple JSON types. Do not publish inaccessible
+project or user directories as a side effect of discovery.
+
+Each issue includes `code`, `clientRowId`, zero-based `rowIndex`, `field`,
+`message`, and optionally `allowedValues` or a limit. Return normalized preview,
+errors, warnings, effective owner, processing choice, and expected entity counts
+where determinable. Humans review this preview; agents use the same response.
+
+Validation records the draft revision, manifest hash and relevant configuration
+version/hash. Commit rechecks authorization and current validation rules. If
+configuration or effective input changed, require a fresh validation rather than
+silently committing a different interpretation. A successful preview cannot
+guarantee later infrastructure or database success.
+
+## Authentication and ownership
+
+Reuse `JwtService` verification with a new submissions filter. Add explicit import
+capability issuance, either an additive opt-in to `AuthToken` or a dedicated
+issuance route sharing its credential verification. Prefer the additive option
+with unchanged behavior when omitted. A capability such as `submissions:write`
+requires current account permission and installation enablement; a client cannot
+self-assert it. No change to read-token behavior or existing search wiring.
+
+Browser session requests use existing identity and CSRF protections as applicable;
+the new writes must enforce CSRF protection when cookie authentication is used.
+Bearer authentication remains stateless. When a Bearer header is present, a bad
+token fails rather than falling back to cookies. Resolve roles against the token
+identity, including mixed cookie/token requests. Do not inherit another session's
+role checks. Recheck account authorization at commit/execution.
+
+Persist owner/context before accepting bytes. Check them on every draft, upload,
+manifest, validation, commit, and result operation. Initially drafts are private
+to their creator plus explicit administrative access. Imported records retain
+existing Wildbook visibility rules. Broader draft sharing is a separate feature.
+
+For the pilot, use dedicated integration accounts and short-lived tokens minted
+by a trusted integration service. Keep passwords out of agent prompts and browser
+third-party apps. Token expiry does not cancel an accepted job; the same identity
+can reauthenticate and resume polling/uploading. Broader third-party delegated
+authorization, revocation UX, and OAuth can follow without changing row intake.
+
+## Upload handling
+
+Start with one-file multipart requests, streamed to staging with bounded memory.
+Reuse path validation, image validation and asset-store handling. Preserve original
+filenames in a manifest; reject duplicate logical names with different content,
+including collisions after filename cleaning or case normalization on the target
+filesystem. Same name and digest is a retry success, not a second file.
+
+Freeze the file manifest at commit. Upload completion must be atomic: partial
+files never become valid references. The server computes size and digest; client
+claims are advisory. Enforce per-file, request, draft, per-user storage, row, and
+active-job limits. Apply media-per-encounter limits after legacy row grouping.
+
+Use owned staging for the new API. Do not expose its drafts through the old
+anonymous upload namespace. Refactor only the small path/storage seams needed to
+pass explicit staged files to the importer. Old upload URLs and destination
+conventions stay stable. If an implementation instead shares directories, it must
+enforce the draft's ownership and freeze across **all** routes that can write
+there; protecting only the new route is insufficient.
+
+Add resumable uploads in a follow-up under the same owned submission. Reuse the
+existing flow chunk engine behind a submission-aware adapter, documenting exact
+chunk geometry and response behavior from code/tests. Scope identifiers to the
+submission and file digest. Do not switch `/upload` or `/ResumableUpload` auth
+chains as a prerequisite for the new feature.
+
+Existing in-memory chunk state is not durable server-restart resume. Stable
+identifiers aid client retries, but cannot restore lost server state. Pin pilot
+uploads to one instance; a multi-instance deployment needs shared state or
+explicit routing plus recovery behavior. Advertise the actual resume guarantee.
+
+## Durable commit and recovery
+
+Add a JDO-backed `Submission` record with owner/context, source, revision, manifest
+reference, immutable payload reference/hash, validation reference, state,
+ImportTask ID, operation key, timestamps, and last error. Store large payloads in
+bounded private storage rather than assuming a giant database JSON field.
+
+Create-request idempotency is scoped to installation, principal, and operation.
+Persist the key and canonical request hash with the new draft under a database
+unique constraint. Same key and input returns the original result; changed input
+returns `409`. Advertise retention and expiration semantics.
+
+For commit, atomically lock the draft, verify its revision and validation, reserve
+one ImportTask ID, freeze the payload/manifest, and persist `queued` before
+returning `202`. A second commit, even with a different key, cannot start a second
+import from that submission. Replaying the original key returns the accepted
+operation before applying stale-revision checks. Keep a durable committed
+tombstone when uploaded staging expires.
+
+Use a bounded worker with database-backed claim/lease state. The worker opens its
+own Shepherd and reloads validated input; do not pass JDO objects across request
+threads. The lifecycle is:
+
+```mermaid
+stateDiagram-v2
+ [*] --> draft
+ draft --> validated: checks pass
+ validated --> draft: rows or files change
+ validated --> queued: atomic commit acceptance
+ queued --> importing: worker claims
+ importing --> imported: database commit recorded
+ importing --> failed: rollback confirmed
+ importing --> needs_reconciliation: outcome uncertain
+ draft --> expired
+ validated --> expired
+```
+
+Record successful import and result IDs in the same database transaction as the
+domain objects where feasible. Preserve existing ImportTask progress transactions
+without treating them as proof that the domain commit succeeded. Keep explicit
+row-to-entity mapping; do not zip legacy result arrays to input rows because rows
+can group into an encounter and caches do not provide that positional contract.
+A narrow optional mapping callback/result in `BulkImporter` is appropriate.
+
+The importer currently launches some post-processing before the caller's final
+commit. For the new adapter, add a narrowly tested opt-in deferred-side-effects
+hook, preserving the legacy default, so work can be scheduled after the commit.
+Persist downstream intent in the commit transaction; reconcile it on restart.
+Do not promise exactly-once IA execution until its dispatch/deduplication boundary
+is demonstrated. Report an uncertain dispatch rather than silently rerunning it.
+
+Queued jobs may be reclaimed safely. An expired lease on an importing job is not
+permission to rerun `createImport`: first establish whether its transaction
+committed and whether the previous worker has stopped. Use a fenced claim for
+any automatic recovery. If outcome cannot be proven, mark
+`needs_reconciliation`, preserve staged files, and require operator recovery.
+This conservative behavior belongs in the first release; unattended full recovery
+can follow. Never describe the current raw background thread as a durable queue.
+
+Filesystem copies and database commits are not one transaction. Track created
+asset paths and clean up unreferenced files after confirmed rollback, with a grace
+period. Do not delete shared/pre-existing media. Expiry cleanup excludes queued,
+active, committed and uncertain jobs until their retention policy permits it.
+
+Return separate import, indexing, detection and identification states. Imported
+records remain imported if IA fails. Unknown downstream state must be explicit;
+an absent status does not mean success. Polling with `Retry-After` is sufficient
+for the MVP; webhooks can follow. Results link to existing task/encounter pages.
+
+Idempotency is scoped to a submission/operation, not global image or observation
+deduplication. The same photograph can legitimately represent multiple encounters.
+For recurring partner feeds, later add a namespaced external-record mapping with
+an explicit conflict/update policy; do not make intake implicitly upsert records.
+
+## Implementation boundary and rollout
+
+1. **Characterize the current path.** Capture object-row/array-row validation,
+ grouping, owner defaults, media handling, task lifecycle, indexing and IA
+ behavior with existing fixtures. Establish frontend/backend test baselines.
+ Finish the OpenAPI examples and agree on the supported MVP field set.
+2. **Add drafts, auth, discovery and simple uploads behind a flag.** Implement
+ ownership, revision control, quotas and validation without enabling commit.
+ No new write authority is granted to existing tokens.
+3. **Add an import execution adapter.** Reuse `BulkImportUtil`, `BulkValidator`,
+ `UploadedFiles.makeMediaAsset` and `BulkImporter`; move only necessary private
+ lifecycle helpers to a small service with explicit user/context/files/options.
+ Separate mechanical extraction from behavior changes in review. Leave the
+ existing servlet's sequencing and defaults covered by characterization tests.
+ Implement durable admission, result mapping and conservative recovery before
+ enabling external commits. Prove post-commit side-effect behavior.
+4. **Pilot one installation and one integration.** Default to 200 rows per job
+ as an operational starting point; publish actual configured limits. Exercise
+ timeouts, crashes and expiry. Observe import latency, failures, quota use,
+ reconciliation cases, search visibility and IA handoffs.
+5. **Expand ergonomics.** Add a reference Python/CLI client and agent instructions
+ from the OpenAPI contract. Humans can use the CLI immediately; a future form
+ can use the same preview/commit endpoints. Add resumable upload, delegated
+ third-party auth and optional callbacks after the core contract is proven.
+
+Do not rewrite `BulkImporter.processRow`, migrate the React UI, replace asset
+storage, or change global location/tolerance defaults as part of this effort.
+The unavoidable new work is the submission lifecycle; call it out in estimates
+rather than presenting it as a few auth-filter changes.
+
+## Acceptance gates
+
+| Area | Required proof |
+| --- | --- |
+| Compatibility | Existing browser import, session/captcha uploads, object/array row payloads, validation errors and task pages retain behavior |
+| Authorization | Read-only JWT rejected; import JWT accepted within authority; cross-owner draft operations denied; cookie CSRF and mixed identities tested |
+| Validation | No domain writes during preview; unknown fields, configured locations, date precision, missing media and grouped media limits enforced |
+| Retry | Lost create/commit responses and concurrent repeated requests create one draft/import; changed keyed payload rejected |
+| Recovery | Crash before dispatch, during import and after domain commit cannot cause blind duplicate imports; uncertain work remains inspectable |
+| Upload | Actual-byte limits, incomplete files, duplicate names/content, normalized-name collisions, freeze and staging cleanup tested |
+| Lifecycle | Imported versus indexed/detected/identified distinguished; IA failure never reports data rollback; row-to-record mapping handles grouping |
+| Persistence | JDO enhancement and clean build; every Shepherd closes; added mappings/constraints tested against PostgreSQL |
+
+Relevant existing tests include `api/bulk/BulkApiPostTest`, `BulkApiOtherTest`,
+`BulkGeneralTest`, `BulkImagesTest`, `BulkImporterMissingAssetTest`, upload path
+and chunk-geometry tests, token-filter/issuance tests, and frontend bulk-import
+and task polling suites. Run meaningful integration/recovery tests in addition
+to those existing suites when implementing; a design-only change needs no build.
+
+Disable new admission to roll back the feature while letting accepted work drain
+or reconcile. Keep legacy imports available. Do not remove new persistence data
+or revoke access to status/results while jobs remain unresolved.
+
+## Accepted product direction — 2026-09-23
+
+The senior engineer accepted these three decisions:
+
+- A sibling submissions API, reusing the bulk-import pipeline.
+- Configured location, strict validation and import-only defaults for the new
+ API, with legacy behavior unchanged. Universal backend location enforcement
+ may be worth a separate correction later; it is not part of this rollout.
+- A pilot with a few approved partners: explicitly enable their accounts for the
+ new API rather than opening access to everyone at launch.
+
+The remaining MVP recommendations (image-backed submissions, polling, simple
+uploads and no deletion of committed records) are detailed above. Before rollout,
+select pilot partners and installation, set byte/storage/concurrency limits and
+retention with operators, and verify the actual deployment topology. Public
+third-party browser authorization remains a later milestone.
diff --git a/docs/design/2026-09-23-submissions-engineer-brief.md b/docs/design/2026-09-23-submissions-engineer-brief.md
new file mode 100644
index 0000000000..cff140a514
--- /dev/null
+++ b/docs/design/2026-09-23-submissions-engineer-brief.md
@@ -0,0 +1,124 @@
+# Generic-client intake: accepted design direction
+
+Based on [your bulk-import proposal](https://qa.wildme.org/bulk-import-next.html)
+and implementation checkout `24cc99aede` (the PR is based on `main` at
+`dcf6f460de`; verification provenance is recorded in the workbench). The senior engineer accepted the three design
+decisions below on 2026-09-23, as relayed by the user. Detailed implementation
+mechanics are documented in the [implementation workbench](submissions/README.md).
+Runtime implementation is locally verified and disabled by default; nothing has
+been deployed. The workbench records test results and remaining QA gates.
+
+## Implementation for engineering review
+
+The sibling resource, durable drafts/uploads, strict validation, commit-once queue,
+worker, result mapping and resumable Python client are implemented locally. All
+six stages received Claude reviews; final rounds report no Critical or Major
+findings. Review transcripts and executed checks are in the workbench above.
+
+The main shared-code change is an opt-in importer mode: it persists through the
+caller's transaction and defers indexing/derivatives. Existing callers keep their
+defaults. This boundary matters because several existing Shepherd creation helpers
+commit internally. New PostgreSQL tests exercise rollback, concurrent acceptance,
+lost commit acknowledgments, recovery holds and actual two-image import mappings.
+
+The pilot intentionally accepts new encounters only, JPEG/PNG uploads and
+import-only processing. It requires configured locations and an explicit account
+allowlist. Uncertain imports require operator reconciliation; search-index
+dispatch is reported without claiming indexing completion. See the
+[pilot runbook](submissions/pilot-runbook.md) for configuration and the remaining
+QA/browser release gate. The Python client provides a human/integration entry
+point; no new browser form is included. The final addition is a public agent skill
+at `/api/v3/agent-skill/submit-sightings`, linked from the existing base toolbox,
+with complete field formats, validation examples and retry guidance.
+
+## Recommendation
+
+Agree with the central approach: reuse the existing bulk-import JSON rows,
+validators, media creation, and importer. Keep the working React workflow and its
+API contract stable. No new spreadsheet parser or encounter-creation pipeline.
+
+I recommend a small **submissions API alongside bulk import** to handle the
+additional lifecycle that agents and third-party integrations need:
+
+**Create draft → upload images → validate → commit once → poll results.**
+
+The new API owns authorization, staged files, validation revisions, and retries.
+The existing importer owns the conversion into Wildbook records. A submission
+can contain one encounter or a batch. Humans can use the same contract through
+a CLI now and a form later.
+
+## Suggested changes to the original proposal
+
+| Topic | Recommendation |
+| --- | --- |
+| Authentication | Reuse JWT infrastructure, but explicitly issue an import capability. A new filter name alone does not distinguish an import token from existing read-only tokens. Keep existing token behavior stable. |
+| API boundary | Add `/api/v3/submissions` for the new lifecycle. Preserve `/api/v3/bulk-import` and browser upload routing. Reuse Java components behind a small adapter. |
+| Retry safety | Persist an owned draft before upload; freeze it at commit; accept one import per submission. Require idempotency keys for creation/commit. A timed-out request must not lead to duplicate encounters. |
+| Upload | Start with streamed, one-file multipart uploads. Use owned staging and a size/digest manifest. Same filename/content can be retried; changed content is a conflict. Add resumable upload afterward using existing chunk mechanics. |
+| Validation | Reuse `BulkValidator` and `BulkImportUtil`. Apply stricter new-API policy at its boundary rather than changing shared defaults immediately. Require configured location, reject unknown fields, and validate media before commit. |
+| Limits | Separate per-file bytes, request bytes, draft storage and row limits. Apply `maximumMediaCountEncounter` to the resulting encounter after row grouping, not to the whole upload batch. |
+| Completion | Report data import separately from indexing, detection and identification. A downstream IA failure must not imply that submitted records disappeared. |
+
+The current checkout has evolved since parts of the proposal: upload requests
+already have configured byte bounds, chunk state keys include the destination
+path, and ImportTask authorization includes collaboration/admin rules. Token TTL
+is server-configured. Implementation should start from these current behaviors.
+
+## Smallest useful first release
+
+- Authenticated, explicitly enrolled integrations; existing browser bulk import
+ continues unchanged.
+- Discovery of supported fields, configured values, processing choices and limits.
+- JSON rows using existing `Class.fieldName` names, plus a client row ID for
+ diagnostics and mapping results back to source records.
+- Simple image uploads; strict validate-and-commit workflow; polling and links to
+ created records and the existing ImportTask.
+- Explicit processing choice: import only, detect, or detect and identify.
+ Recommend import-only as the new API default, preserving old API defaults.
+- Durable commit acceptance and conservative recovery. If a worker crashes and
+ commit outcome is uncertain, expose a reconciliation state instead of blindly
+ retrying record creation.
+
+Defer anonymous intake, general upserts, webhooks, CSV/XLSX parsing, new human UI,
+and broad third-party OAuth. Do not give an agent a user's password; use a trusted
+client/service to obtain its short-lived credential.
+
+## Engineering boundary and rollout
+
+The importer is reusable, but its servlet orchestration is not already a service
+interface. Extract only needed lifecycle helpers, preserving existing defaults
+and sequencing with characterization tests. `BulkImporter` also starts some
+indexing/media-child work before its caller commits; a new durable worker must
+account for that with a small, tested deferred-side-effects seam.
+
+Implement in three reviewable stages:
+
+1. Characterize the existing flow and agree on the new contract.
+2. Add gated authentication, owned drafts, simple uploads and validation.
+3. Add the execution adapter, commit deduplication and recovery tests; pilot one
+ installation/integration before broader enablement.
+
+This is more work than enabling Bearer authentication on the existing POST, but
+the additional work addresses unattended-client behavior without a broad importer
+rewrite. If delivery must be reduced, cut resumable upload and client conveniences
+first; retain ownership and safe commit retry semantics.
+
+## Accepted decisions — 2026-09-23
+
+1. **Use a sibling submissions resource.** Reuse the bulk-import pipeline behind
+ the new lifecycle API.
+2. **Require configured location and strict validation; default to import-only.**
+ Apply these defaults to the new API and preserve legacy behavior. Requiring
+ location universally may be a reasonable later correction, but is outside
+ this rollout to avoid disrupting working imports.
+3. **Pilot with a few approved partners.** Only accounts explicitly enabled for
+ the pilot can use the new API initially. This is what the proposed enrollment
+ allowlist means; the API is not opened to all users at launch.
+
+The [implementation plan](../plans/2026-09-23-submissions-api-implementation.md)
+breaks delivery into six reviewable changes, starting with the OpenAPI contract
+and compatibility tests, then gated drafts, uploads and validation. Pilot partners,
+installation, resource limits and retention need operational selection before rollout.
+
+The [supporting design](2026-09-23-submissions-api.md) includes proposed endpoints,
+payloads, state transitions, recovery behavior and regression acceptance gates.
diff --git a/docs/design/submissions/README.md b/docs/design/submissions/README.md
new file mode 100644
index 0000000000..306fadafed
--- /dev/null
+++ b/docs/design/submissions/README.md
@@ -0,0 +1,148 @@
+# Submissions contract workbench
+
+This folder contains the local implementation contract, review evidence and pilot
+handoff. The API has not been deployed. The accepted direction and
+implementation sequence are in [the implementation plan](../../plans/2026-09-23-submissions-api-implementation.md).
+
+- `openapi.yaml`: version-one contract; published OpenAPI reflects the implemented pilot subset.
+- `examples.json`: example envelopes and structured validation issues.
+- `scripts/submissions/check_contract.py` (repository root): checks local references,
+ example schema validity and key routing/precondition invariants. Requires PyYAML
+ and jsonschema; it is not a complete OpenAPI conformance validator.
+- `BulkSubmissionCompatibilityTest`: new characterization of legacy optional
+ location, year-only dates, equivalent row encodings and unknown-field policy.
+
+Run the local contract checks from the repository root:
+
+```bash
+python3 scripts/submissions/check_contract.py
+```
+
+Stage-one completion requires the targeted legacy baseline, new characterization
+tests and Claude review. Each later implementation stage also requires Claude
+review before proceeding. Review findings and dispositions are recorded under reviews/.
+
+## Review status
+
+The user explicitly approved sending relevant project files to Claude for
+read-only reviews at every stage, excluding credentials and unrelated material.
+Stages 1–6 have converged with no Critical or Major findings after corrections.
+Claude reviewed each stage read-only; execution evidence below comes from local checks.
+Full review transcripts and dispositions are under reviews/.
+
+## Local verification
+
+Baseline on implementation checkout `24cc99aede`, before runtime changes
+(PR base is `dcf6f460de`; see the provenance note below):
+
+```bash
+mvn -o test -Dtest=BulkApiPostTest,BulkApiOtherTest,BulkGeneralTest,BulkImagesTest,BulkImporterMissingAssetTest,AuthTokenTest,AuthTokenStepUpTest,WildbookTokenAuthenticationFilterTest,'UploadPaths*Test'
+```
+
+Result: **106 tests, 0 failures, 0 errors, 0 skipped; BUILD SUCCESS**.
+This is the selected unit-test baseline, not a full build or database recovery test.
+The run emitted background datastore diagnostics from existing mocked importer
+fixtures but completed successfully. Maven needed execution outside the sandbox
+because the canonical capitalized checkout path was treated as read-only.
+
+Initial stage-one contract check (historical): **10 operations/seven examples passed**.
+The current contract check covers 11 operations and 18 examples.
+
+New characterization suite:
+
+```bash
+mvn -o test -Dtest=BulkSubmissionCompatibilityTest
+```
+
+Result: **4 tests, 0 failures, 0 errors, 0 skipped; BUILD SUCCESS**.
+The new suite ran separately after the baseline, before runtime implementation.
+
+After Claude round-one fixes, the contract check passes 11 operations and 16
+positive/negative examples. The strengthened tests pass:
+
+```bash
+mvn -o test -Dtest=BulkSubmissionCompatibilityTest,BulkApiPostTest
+```
+
+**21 tests, 0 failures, 0 errors, 0 skipped; BUILD SUCCESS.**
+
+
+## Implementation verification
+
+Stage 2 authentication/draft/persistence tests: **38 passed**, including PostgreSQL
+restart, concurrent create/edit, quota race and rollback checks.
+
+Latest combined upload/validation/importer/queue run: **37 passed**, zero failures,
+errors or skips. Command:
+
+```bash
+mvn -o test -Dtest=SubmissionFilesTest,SubmissionValidatorTest,SubmissionStoreDbTest,BulkImporterSubmissionBoundaryTest,BulkSubmissionCompatibilityTest,BulkApiPostTest,BulkImporterMissingAssetTest
+```
+
+Client: `python3 -m unittest discover -s scripts/submissions -p test_client.py`:
+**7 passed**, including command-flow create/commit recovery, row correction,
+pagination and state-lock tests. Contract checker: **11 operations and 18 examples passed**.
+
+The full clean build ran the frontend: **21 bulk-import suites passed**; the entire
+frontend had **130 suites passed, 16 failed; 1,362 tests passed, 40 failed**.
+Failures are in unchanged frontend sources (including a missing Citation module
+and existing component test expectations); no base-commit frontend comparison was
+run, so these are not asserted to be proven pre-existing failures. Production
+frontend compilation completed. The clean build's first Java run had **1,108
+tests, one failure, one error, seven skipped**. Both failures were new test
+fixtures missing usernames. Those fixtures are corrected; the final full Java
+rerun passed as recorded below.
+
+Corrected PostgreSQL rerun: `mvn -o test -Dtest=SubmissionStoreDbTest`:
+**15 passed, zero failures/errors/skips; BUILD SUCCESS**. This includes the actual
+two-image adapter import, caller rollback, cleanup retention/locking, daily quota,
+invalid-validation rejection and replay pagination.
+
+`check_contract.py --runtime` passes both captured server responses (capabilities
+and commit acceptance) against both the draft and published OpenAPI schemas.
+These local tests do not replace the QA/browser release gate in
+[pilot-runbook.md](pilot-runbook.md).
+No installation has been deployed or enrolled.
+
+Final full Java regression: `mvn -o install`: **1,109 tests, zero failures, zero
+errors, seven skipped**. This includes all submissions/authentication tests and
+the existing bulk-import, upload, permissions and database suites. **BUILD SUCCESS**,
+including WAR packaging and local Maven installation. Frontend tests were run by the preceding clean invocation and
+retain the separate failure limitation above.
+
+Existing published OpenAPI paths and schemas were compared with the original
+checkout and remain unchanged; new route/auth documentation is additive.
+
+## Final addition: published agent skill
+
+At the user's request, added after the full implementation/build: the public
+`/api/v3/agent-skill/submit-sightings` resource, registered in AgentSkill and linked
+from the base toolbox and read-only API reference. It includes all supported
+fields, installation settings discovery, wire formats, field-specific validation
+failures, authorization boundaries, retry reconciliation and result mapping.
+`mvn -o test -Dtest=AgentSkillTest,AgentSkillContentTest`: **15 passed, zero
+failures/errors/skips; BUILD SUCCESS**. Existing skill routing/content checks and
+new runtime-parsed request examples passed. Two Claude review rounds converged
+with no Critical or Major issues; transcripts and disposition are under reviews/.
+This addition does not change submissions runtime behavior.
+
+Final artifact after the agent-skill addition: `mvn -o -DskipTests package`: **BUILD
+SUCCESS**. Tests were deliberately not rerun during packaging; the full API and
+subsequent skill test runs are recorded above. Verified the WAR contains byte-for-byte
+current submit-sightings, toolbox and API-reference resources, the new catalog
+registration, submissions servlet and JDO metadata. Artifact:
+`target/wildbook-10.14.war`. The deployment descriptor is handled separately as
+explained in the pilot runbook. No deployment or account enrollment was performed.
+
+## PR base and verification provenance
+
+The PR branch is based on `main` at `dcf6f460de`, excluding the separate mobile-layout
+commit `24cc99aede` present during implementation/testing. The difference between
+those bases contains only frontend files; Java sources and dependencies are identical.
+The frontend results and built WAR above therefore describe the earlier checkout,
+not a fresh frontend build of the PR base. No frontend changes are part of this PR.
+QA/browser verification on the final branch remains a release gate.
+
+Claude also approved the final PR handoff with no blockers; see
+[PR handoff review](reviews/pr-handoff-review.md). The suggested link and
+verification-provenance wording clarifications were incorporated.
diff --git a/docs/design/submissions/examples.json b/docs/design/submissions/examples.json
new file mode 100644
index 0000000000..654bcf6834
--- /dev/null
+++ b/docs/design/submissions/examples.json
@@ -0,0 +1,292 @@
+{
+ "create": {
+ "schema": "Create",
+ "value": {
+ "contractVersion": "1",
+ "source": {
+ "name": "pilot-client",
+ "batchId": "survey-001"
+ }
+ }
+ },
+ "rows": {
+ "schema": "Rows",
+ "value": {
+ "rows": [
+ {
+ "clientRowId": "observation-1",
+ "fields": {
+ "Encounter.year": 2026,
+ "Encounter.locationID": "configured-location",
+ "Encounter.genus": "Loxodonta",
+ "Encounter.specificEpithet": "africana",
+ "Encounter.mediaAsset0": "observation-1.jpg"
+ }
+ }
+ ]
+ }
+ },
+ "invalidLocation": {
+ "schema": "Issue",
+ "value": {
+ "code": "LOCATION_NOT_CONFIGURED",
+ "message": "Choose a configured location.",
+ "clientRowId": "observation-1",
+ "rowIndex": 0,
+ "field": "Encounter.locationID"
+ }
+ },
+ "draft": {
+ "schema": "Submission",
+ "value": {
+ "id": "00000000-0000-4000-8000-000000000001",
+ "contractVersion": "1",
+ "revision": 0,
+ "state": "draft",
+ "source": {
+ "name": "pilot-client"
+ },
+ "processing": {
+ "mode": "import-only"
+ },
+ "createdAt": "2026-09-23T20:00:00Z",
+ "expiresAt": "2026-09-30T20:00:00Z",
+ "rowCount": 0
+ }
+ },
+ "accepted": {
+ "schema": "Accepted",
+ "value": {
+ "submissionId": "00000000-0000-4000-8000-000000000001",
+ "operationId": "00000000-0000-4000-8000-000000000002",
+ "importTaskId": "00000000-0000-4000-8000-000000000003",
+ "acceptedRevision": 2,
+ "statusUrl": "/api/v3/submissions/00000000-0000-4000-8000-000000000001"
+ }
+ },
+ "validationFailed": {
+ "schema": "Validation",
+ "value": {
+ "id": "00000000-0000-4000-8000-000000000004",
+ "submissionId": "00000000-0000-4000-8000-000000000001",
+ "revision": 2,
+ "valid": false,
+ "configDigest": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
+ "manifestDigest": "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb",
+ "errors": [
+ {
+ "code": "LOCATION_NOT_CONFIGURED",
+ "message": "Choose a configured location.",
+ "clientRowId": "observation-1",
+ "rowIndex": 0,
+ "field": "Encounter.locationID"
+ }
+ ],
+ "warnings": [],
+ "normalizedRows": [],
+ "effectiveOwnerId": "00000000-0000-4000-8000-000000000005",
+ "processing": {
+ "mode": "import-only"
+ }
+ }
+ },
+ "importedResults": {
+ "schema": "Results",
+ "value": {
+ "submissionId": "00000000-0000-4000-8000-000000000001",
+ "state": "imported",
+ "indexing": {
+ "state": "pending"
+ },
+ "detection": {
+ "state": "skipped"
+ },
+ "identification": {
+ "state": "skipped"
+ },
+ "rows": [
+ {
+ "clientRowId": "observation-1",
+ "encounterIds": [
+ "00000000-0000-4000-8000-000000000006"
+ ],
+ "occurrenceIds": [],
+ "individualIds": [],
+ "mediaAssetIds": [
+ 100
+ ]
+ }
+ ],
+ "errors": []
+ }
+ },
+ "manifest": {
+ "schema": "Manifest",
+ "value": {
+ "submissionId": "00000000-0000-4000-8000-000000000001",
+ "revision": 1,
+ "files": [
+ {
+ "name": "observation-1.jpg",
+ "sizeBytes": 1024,
+ "sha256": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
+ "state": "complete",
+ "mediaType": "image/jpeg"
+ }
+ ]
+ }
+ },
+ "validationPassed": {
+ "schema": "Validation",
+ "value": {
+ "id": "00000000-0000-4000-8000-000000000004",
+ "submissionId": "00000000-0000-4000-8000-000000000001",
+ "revision": 2,
+ "valid": true,
+ "configDigest": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
+ "manifestDigest": "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb",
+ "errors": [],
+ "warnings": [],
+ "normalizedRows": [
+ {
+ "clientRowId": "observation-1",
+ "fields": {
+ "Encounter.year": 2026,
+ "Encounter.locationID": "configured-location",
+ "Encounter.genus": "Loxodonta",
+ "Encounter.specificEpithet": "africana",
+ "Encounter.mediaAsset0": "observation-1.jpg"
+ }
+ }
+ ],
+ "effectiveOwnerId": "00000000-0000-4000-8000-000000000005",
+ "processing": {
+ "mode": "import-only"
+ }
+ }
+ },
+ "error409": {
+ "schema": "Error",
+ "value": {
+ "code": "IDEMPOTENCY_KEY_REUSED",
+ "message": "Example IDEMPOTENCY_KEY_REUSED",
+ "requestId": "req-1",
+ "issues": []
+ }
+ },
+ "error412": {
+ "schema": "Error",
+ "value": {
+ "code": "REVISION_STALE",
+ "message": "Example REVISION_STALE",
+ "requestId": "req-1",
+ "issues": []
+ }
+ },
+ "error428": {
+ "schema": "Error",
+ "value": {
+ "code": "PRECONDITION_REQUIRED",
+ "message": "Example PRECONDITION_REQUIRED",
+ "requestId": "req-1",
+ "issues": []
+ }
+ },
+ "error429": {
+ "schema": "Error",
+ "value": {
+ "code": "LIMIT_EXCEEDED",
+ "message": "Example LIMIT_EXCEEDED",
+ "requestId": "req-1",
+ "issues": []
+ }
+ },
+ "capabilities": {
+ "schema": "Capabilities",
+ "value": {
+ "contractVersion": "1",
+ "admissionEnabled": true,
+ "commitEnabled": false,
+ "processingModes": [
+ "import-only"
+ ],
+ "authentication": [
+ "bearer"
+ ],
+ "limits": {
+ "maxRows": 200,
+ "maxFileBytes": 200,
+ "maxRequestBytes": 200,
+ "maxDraftBytes": 200,
+ "maxMediaPerEncounter": 200,
+ "maxActiveJobs": 200,
+ "draftTtlSeconds": 200,
+ "idempotencyRetentionSeconds": 200
+ },
+ "rowFields": {}
+ }
+ },
+ "rejectNullField": {
+ "schema": "Rows",
+ "valid": false,
+ "value": {
+ "rows": [
+ {
+ "clientRowId": "a",
+ "fields": {
+ "Encounter.year": null
+ }
+ }
+ ]
+ }
+ },
+ "rejectUnknownCreateProperty": {
+ "schema": "Create",
+ "valid": false,
+ "value": {
+ "contractVersion": "1",
+ "source": {
+ "name": "test"
+ },
+ "ownerId": "override"
+ }
+ },
+ "rejectBadUuid": {
+ "schema": "Submission",
+ "valid": false,
+ "value": {
+ "id": "not-a-uuid",
+ "contractVersion": "1",
+ "revision": 0,
+ "state": "draft",
+ "source": {
+ "name": "pilot-client"
+ },
+ "processing": {
+ "mode": "import-only"
+ },
+ "createdAt": "2026-09-23T20:00:00Z",
+ "expiresAt": "2026-09-30T20:00:00Z",
+ "rowCount": 0
+ }
+ },
+ "rejectBadTimestamp": {
+ "schema": "Submission",
+ "valid": false,
+ "value": {
+ "id": "00000000-0000-4000-8000-000000000001",
+ "contractVersion": "1",
+ "revision": 0,
+ "state": "draft",
+ "source": {
+ "name": "pilot-client"
+ },
+ "processing": {
+ "mode": "import-only"
+ },
+ "createdAt": "not-a-date",
+ "expiresAt": "2026-09-30T20:00:00Z",
+ "rowCount": 0
+ }
+ }
+}
diff --git a/docs/design/submissions/openapi.yaml b/docs/design/submissions/openapi.yaml
new file mode 100644
index 0000000000..adf72e9bfe
--- /dev/null
+++ b/docs/design/submissions/openapi.yaml
@@ -0,0 +1,1588 @@
+openapi: 3.0.3
+info:
+ title: Wildbook Submissions API (draft, not deployed)
+ version: 1.0.0-draft
+ description: 'Sibling intake API. Private enrolled-partner pilot; import-only by
+ default. Existing bulk-import endpoints are unchanged. Bearer credentials must
+ contain an explicitly issued submissions capability; existing identity-only tokens
+ do not authorize intake. Session writes require CSRF protection when enabled.
+ All limits and retention are installation-configured. Responses carrying private
+ draft data use Cache-Control: no-store. Relevant configuration and permissions
+ are rechecked before execution.'
+servers:
+- url: /
+security:
+- submissionBearer: []
+paths:
+ /api/v3/submissions/capabilities:
+ get:
+ operationId: submissionCapabilities
+ description: Returns limits and implemented capabilities for this installation
+ and principal.
+ parameters: []
+ responses:
+ '200':
+ description: Success
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Capabilities'
+ '400':
+ description: Malformed request
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '401':
+ description: Authentication required or invalid
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '403':
+ description: Not enrolled or not permitted
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '429':
+ description: Quota or concurrency limit
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ headers:
+ Retry-After:
+ schema:
+ type: integer
+ minimum: 1
+ '503':
+ description: Feature temporarily unavailable
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '500':
+ description: Internal failure; retry only according to operation idempotency
+ rules
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '405':
+ description: Method not allowed; Allow header lists supported methods
+ /api/v3/submissions:
+ post:
+ operationId: createSubmission
+ description: 'Creates a private draft. Replays return the original 201 body
+ and resource URL. Omitted processing is import-only. The replayed body/ETag
+ may be old: GET the resource before any mutation.'
+ parameters:
+ - $ref: '#/components/parameters/IdempotencyKey'
+ responses:
+ '201':
+ description: Success
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Submission'
+ headers:
+ ETag:
+ schema:
+ type: string
+ description: Quoted current revision.
+ Location:
+ schema:
+ type: string
+ description: Resource or status URL.
+ '400':
+ description: Malformed request
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '401':
+ description: Authentication required or invalid
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '403':
+ description: Not enrolled or not permitted
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '429':
+ description: Quota or concurrency limit
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ headers:
+ Retry-After:
+ schema:
+ type: integer
+ minimum: 1
+ '503':
+ description: Feature temporarily unavailable
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '409':
+ description: State, key or content conflict
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '413':
+ description: Input too large
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '422':
+ description: Input or validation unacceptable
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '500':
+ description: Internal failure; retry only according to operation idempotency
+ rules
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '405':
+ description: Method not allowed; Allow header lists supported methods
+ requestBody:
+ required: true
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Create'
+ /api/v3/submissions/{id}:
+ get:
+ operationId: getSubmission
+ description: Returns the current durable status; queued and in-flight jobs remain
+ inspectable when new admission is disabled. Owners receive 200 with cancelled
+ or expired state during tombstone retention; 404 after final purge.
+ parameters:
+ - $ref: '#/components/parameters/SubmissionId'
+ responses:
+ '200':
+ description: Success
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Submission'
+ headers:
+ ETag:
+ schema:
+ type: string
+ description: Quoted current revision.
+ '400':
+ description: Malformed request
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '401':
+ description: Authentication required or invalid
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '403':
+ description: Not enrolled or not permitted
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '404':
+ description: Not found or not visible
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '429':
+ description: Quota or concurrency limit
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ headers:
+ Retry-After:
+ schema:
+ type: integer
+ minimum: 1
+ '503':
+ description: Feature temporarily unavailable
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '500':
+ description: Internal failure; retry only according to operation idempotency
+ rules
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '405':
+ description: Method not allowed; Allow header lists supported methods
+ delete:
+ operationId: cancelSubmission
+ description: Cancels only draft or validated state. After owner checks, already-cancelled
+ state returns 204 regardless of the supplied If-Match (even if stale). The
+ If-Match header is always required, but its value is not compared on repeated
+ cancellation. All queued/importing/imported/failed/needs_reconciliation/expired
+ states return 409. This never deletes biological records.
+ parameters:
+ - $ref: '#/components/parameters/SubmissionId'
+ - $ref: '#/components/parameters/IfMatch'
+ responses:
+ '204':
+ description: Success
+ '400':
+ description: Malformed request
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '401':
+ description: Authentication required or invalid
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '403':
+ description: Not enrolled or not permitted
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '404':
+ description: Not found or not visible
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '429':
+ description: Quota or concurrency limit
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ headers:
+ Retry-After:
+ schema:
+ type: integer
+ minimum: 1
+ '503':
+ description: Feature temporarily unavailable
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '409':
+ description: State, key or content conflict
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '412':
+ description: Stale revision
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '428':
+ description: Required revision precondition missing
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '500':
+ description: Internal failure; retry only according to operation idempotency
+ rules
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '405':
+ description: Method not allowed; Allow header lists supported methods
+ /api/v3/submissions/{id}/rows:
+ put:
+ operationId: replaceSubmissionRows
+ description: Replaces all rows; unique clientRowId values required. Invalidates
+ prior validation.
+ parameters:
+ - $ref: '#/components/parameters/SubmissionId'
+ - $ref: '#/components/parameters/IfMatch'
+ responses:
+ '200':
+ description: Success
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Submission'
+ headers:
+ ETag:
+ schema:
+ type: string
+ description: Quoted current revision.
+ '400':
+ description: Malformed request
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '401':
+ description: Authentication required or invalid
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '403':
+ description: Not enrolled or not permitted
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '404':
+ description: Not found or not visible
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '429':
+ description: Quota or concurrency limit
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ headers:
+ Retry-After:
+ schema:
+ type: integer
+ minimum: 1
+ '503':
+ description: Feature temporarily unavailable
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '409':
+ description: State, key or content conflict
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '412':
+ description: Stale revision
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '413':
+ description: Input too large
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '422':
+ description: Input or validation unacceptable
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '428':
+ description: Required revision precondition missing
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '500':
+ description: Internal failure; retry only according to operation idempotency
+ rules
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '405':
+ description: Method not allowed; Allow header lists supported methods
+ requestBody:
+ required: true
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Rows'
+ get:
+ operationId: getSubmissionRows
+ description: Returns rows as accepted; compare by JSON value equality after
+ a lost PUT response. GET the current ETag before another mutation. Rows remain
+ readable during cancelled/expired tombstone retention.
+ parameters:
+ - $ref: '#/components/parameters/SubmissionId'
+ responses:
+ '200':
+ description: Stored rows, or empty array before first PUT
+ headers:
+ ETag:
+ schema:
+ type: string
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/StoredRows'
+ '400':
+ description: Malformed request
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '401':
+ description: Authentication required or invalid
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '403':
+ description: Not enrolled or not permitted
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '404':
+ description: Not found or not visible
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '429':
+ description: Quota or concurrency limit
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ headers:
+ Retry-After:
+ schema:
+ type: integer
+ minimum: 1
+ '503':
+ description: Feature temporarily unavailable
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '500':
+ description: Internal failure; retry only according to operation idempotency
+ rules
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '405':
+ description: Method not allowed; Allow header lists supported methods
+ /api/v3/submissions/{id}/files:
+ get:
+ operationId: getSubmissionFiles
+ description: Complete staged images only.
+ parameters:
+ - $ref: '#/components/parameters/SubmissionId'
+ responses:
+ '200':
+ description: Success
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Manifest'
+ headers:
+ ETag:
+ schema:
+ type: string
+ description: Quoted current revision.
+ '400':
+ description: Malformed request
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '401':
+ description: Authentication required or invalid
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '403':
+ description: Not enrolled or not permitted
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '404':
+ description: Not found or not visible
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '429':
+ description: Quota or concurrency limit
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ headers:
+ Retry-After:
+ schema:
+ type: integer
+ minimum: 1
+ '503':
+ description: Feature temporarily unavailable
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '410':
+ description: Expired or cancelled draft during tombstone retention
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '500':
+ description: Internal failure; retry only according to operation idempotency
+ rules
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '405':
+ description: Method not allowed; Allow header lists supported methods
+ post:
+ operationId: uploadSubmissionFile
+ description: 'One image per request. Server computes digest. Serialize uploads;
+ after a lost response GET manifest and reconcile name/digest and ETag. Same
+ name/content at current revision is a retry success; differing content is
+ 409. Incomplete files are never visible. File.name equals the original multipart
+ filename used in Encounter.mediaAssetN. Reject unsafe names or names requiring
+ cleaning; never silently rename. Same-content retries do not advance revision.
+ Pilot: JPEG/PNG, matching extension, max 200 files and 200 MiB completed bytes
+ per draft; exceeding either returns 413. At most one additional bounded retry
+ candidate exists during upload. Staging must be configured. Serialize uploads;
+ processing contention returns 429.'
+ parameters:
+ - $ref: '#/components/parameters/SubmissionId'
+ - $ref: '#/components/parameters/IfMatch'
+ responses:
+ '200':
+ description: Success
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Manifest'
+ headers:
+ ETag:
+ schema:
+ type: string
+ description: Quoted current revision.
+ '400':
+ description: Malformed request
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '401':
+ description: Authentication required or invalid
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '403':
+ description: Not enrolled or not permitted
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '404':
+ description: Not found or not visible
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '429':
+ description: Quota or concurrency limit
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ headers:
+ Retry-After:
+ schema:
+ type: integer
+ minimum: 1
+ '503':
+ description: Feature temporarily unavailable
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '409':
+ description: State, key or content conflict
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '412':
+ description: Stale revision
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '413':
+ description: Input too large
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '422':
+ description: Input or validation unacceptable
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '428':
+ description: Required revision precondition missing
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '500':
+ description: Internal failure; retry only according to operation idempotency
+ rules
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '405':
+ description: Method not allowed; Allow header lists supported methods
+ '408':
+ description: Upload exceeded wall-clock limit
+ requestBody:
+ required: true
+ content:
+ multipart/form-data:
+ schema:
+ type: object
+ additionalProperties: false
+ required:
+ - file
+ properties:
+ file:
+ type: string
+ format: binary
+ /api/v3/submissions/{id}/validate:
+ post:
+ operationId: validateSubmission
+ description: Validate current revision without changing it. Validation failures,
+ including empty rows, return 200 with valid=false and source-row issues. No
+ domain objects are created.
+ parameters:
+ - $ref: '#/components/parameters/SubmissionId'
+ - $ref: '#/components/parameters/IfMatch'
+ responses:
+ '200':
+ description: Success
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Validation'
+ headers:
+ ETag:
+ schema:
+ type: string
+ description: Unchanged quoted input revision; use on commit.
+ '400':
+ description: Malformed request
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '401':
+ description: Authentication required or invalid
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '403':
+ description: Not enrolled or not permitted
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '404':
+ description: Not found or not visible
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '429':
+ description: Quota or concurrency limit
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ headers:
+ Retry-After:
+ schema:
+ type: integer
+ minimum: 1
+ '503':
+ description: Feature temporarily unavailable
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '409':
+ description: State, key or content conflict
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '412':
+ description: Stale revision
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '413':
+ description: Input too large
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '428':
+ description: Required revision precondition missing
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '500':
+ description: Internal failure; retry only according to operation idempotency
+ rules
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '405':
+ description: Method not allowed; Allow header lists supported methods
+ requestBody:
+ required: true
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Validate'
+ /api/v3/submissions/{id}/commit:
+ post:
+ operationId: commitSubmission
+ description: Atomically freezes a validated draft and accepts at most one execution.
+ Same keyed request returns the original acceptance before checking stale If-Match.
+ A different key cannot queue a second execution. Invalid validation returns
+ 422 VALIDATION_INVALID; stale validation/config returns 409 VALIDATION_STALE;
+ a different key after acceptance returns 409 ALREADY_COMMITTED. After admission
+ shutdown recover via GET status/results; mutation retries may return 503 instead
+ of replaying 202.
+ parameters:
+ - $ref: '#/components/parameters/SubmissionId'
+ - $ref: '#/components/parameters/IfMatch'
+ - $ref: '#/components/parameters/IdempotencyKey'
+ responses:
+ '202':
+ description: Success
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Accepted'
+ headers:
+ Location:
+ schema:
+ type: string
+ description: Resource or status URL.
+ '400':
+ description: Malformed request
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '401':
+ description: Authentication required or invalid
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '403':
+ description: Not enrolled or not permitted
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '404':
+ description: Not found or not visible
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '429':
+ description: Quota or concurrency limit
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ headers:
+ Retry-After:
+ schema:
+ type: integer
+ minimum: 1
+ '503':
+ description: Feature temporarily unavailable
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '409':
+ description: State, key or content conflict
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '412':
+ description: Stale revision
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '413':
+ description: Input too large
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '422':
+ description: Input or validation unacceptable
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '428':
+ description: Required revision precondition missing
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '500':
+ description: Internal failure; retry only according to operation idempotency
+ rules
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '405':
+ description: Method not allowed; Allow header lists supported methods
+ requestBody:
+ required: true
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Commit'
+ /api/v3/submissions/{id}/results:
+ get:
+ operationId: getSubmissionResults
+ description: Stable row-order pagination; records may share entities. Downstream
+ failure does not undo successful import. Before results exist returns empty
+ rows and current state.
+ parameters:
+ - $ref: '#/components/parameters/SubmissionId'
+ - name: cursor
+ in: query
+ schema:
+ type: string
+ minLength: 1
+ - name: limit
+ in: query
+ schema:
+ type: integer
+ minimum: 1
+ maximum: 200
+ default: 100
+ responses:
+ '200':
+ description: Success
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Results'
+ '400':
+ description: Malformed request
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '401':
+ description: Authentication required or invalid
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '403':
+ description: Not enrolled or not permitted
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '404':
+ description: Not found or not visible
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '429':
+ description: Quota or concurrency limit
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ headers:
+ Retry-After:
+ schema:
+ type: integer
+ minimum: 1
+ '503':
+ description: Feature temporarily unavailable
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '500':
+ description: Internal failure; retry only according to operation idempotency
+ rules
+ content:
+ application/json:
+ schema:
+ $ref: '#/components/schemas/Error'
+ '405':
+ description: Method not allowed; Allow header lists supported methods
+components:
+ securitySchemes:
+ submissionBearer:
+ type: http
+ scheme: bearer
+ bearerFormat: JWT
+ description: Signed submissions capability and currently enrolled account required
+ for mutations; owner access retained for status after unenrollment.
+ parameters:
+ SubmissionId:
+ name: id
+ in: path
+ required: true
+ schema:
+ type: string
+ format: uuid
+ IfMatch:
+ name: If-Match
+ in: header
+ required: true
+ schema:
+ type: string
+ pattern: ^"[0-9]{1,18}"$
+ description: Quoted current revision; 428 when absent, 412 when stale.
+ IdempotencyKey:
+ name: Idempotency-Key
+ in: header
+ required: true
+ schema:
+ type: string
+ minLength: 1
+ maxLength: 128
+ description: Scoped to context, authenticated principal and operation. Same
+ input replays the original response; changed input is 409.
+ schemas:
+ Error:
+ type: object
+ additionalProperties: true
+ required:
+ - code
+ - message
+ - requestId
+ properties:
+ code:
+ type: string
+ enum:
+ - BAD_REQUEST
+ - AUTHENTICATION_REQUIRED
+ - ACCESS_DENIED
+ - NOT_FOUND
+ - GONE
+ - PRECONDITION_REQUIRED
+ - REVISION_STALE
+ - IDEMPOTENCY_KEY_REUSED
+ - ALREADY_COMMITTED
+ - VALIDATION_STALE
+ - VALIDATION_INVALID
+ - INVALID_STATE
+ - LIMIT_EXCEEDED
+ - ADMISSION_DISABLED
+ - DUPLICATE_CLIENT_ROW_ID
+ - FILE_CONTENT_CONFLICT
+ - INTERNAL_ERROR
+ - CAPABILITY_UNAVAILABLE
+ message:
+ type: string
+ minLength: 1
+ requestId:
+ type: string
+ minLength: 1
+ issues:
+ type: array
+ items:
+ $ref: '#/components/schemas/Issue'
+ description: DUPLICATE_CLIENT_ROW_ID is 422; INVALID_STATE is 409; PRECONDITION_REQUIRED
+ is 428; REVISION_STALE is 412; IDEMPOTENCY_KEY_REUSED and ALREADY_COMMITTED
+ are 409; LIMIT_EXCEEDED is 413 for input size or 429 for quotas; ADMISSION_DISABLED
+ is 503.
+ Source:
+ type: object
+ additionalProperties: false
+ required:
+ - name
+ properties:
+ name:
+ type: string
+ minLength: 1
+ maxLength: 128
+ batchId:
+ type: string
+ minLength: 1
+ maxLength: 256
+ Processing:
+ type: object
+ additionalProperties: false
+ required:
+ - mode
+ properties:
+ mode:
+ type: string
+ enum:
+ - import-only
+ - detect
+ - detect-and-identify
+ default: import-only
+ description: Omit the entire processing object to select import-only. When provided,
+ mode is required.
+ Create:
+ type: object
+ additionalProperties: false
+ required:
+ - contractVersion
+ - source
+ properties:
+ contractVersion:
+ type: string
+ enum:
+ - '1'
+ source:
+ $ref: '#/components/schemas/Source'
+ processing:
+ $ref: '#/components/schemas/Processing'
+ Row:
+ type: object
+ additionalProperties: false
+ required:
+ - clientRowId
+ - fields
+ properties:
+ clientRowId:
+ type: string
+ minLength: 1
+ maxLength: 128
+ fields:
+ type: object
+ minProperties: 1
+ additionalProperties:
+ oneOf:
+ - type: string
+ - type: number
+ - type: boolean
+ description: Normalized bulk-import field names. Actual supported fields,
+ required values and types are supplied by capabilities. Nulls and unknown
+ fields are rejected in this contract.
+ maxProperties: 256
+ Rows:
+ type: object
+ additionalProperties: false
+ required:
+ - rows
+ properties:
+ rows:
+ type: array
+ items:
+ $ref: '#/components/schemas/Row'
+ minItems: 1
+ maxItems: 200
+ Submission:
+ type: object
+ additionalProperties: true
+ required:
+ - id
+ - contractVersion
+ - revision
+ - state
+ - source
+ - processing
+ - createdAt
+ - expiresAt
+ - rowCount
+ properties:
+ id:
+ type: string
+ format: uuid
+ contractVersion:
+ type: string
+ enum:
+ - '1'
+ revision:
+ type: integer
+ minimum: 0
+ state:
+ type: string
+ enum:
+ - draft
+ - validated
+ - queued
+ - importing
+ - imported
+ - failed
+ - needs_reconciliation
+ - cancelled
+ - expired
+ source:
+ $ref: '#/components/schemas/Source'
+ processing:
+ $ref: '#/components/schemas/Processing'
+ createdAt:
+ type: string
+ format: date-time
+ expiresAt:
+ type: string
+ format: date-time
+ importTaskId:
+ type: string
+ format: uuid
+ validationId:
+ type: string
+ format: uuid
+ rowsDigest:
+ type: string
+ pattern: ^[a-f0-9]{64}$
+ description: Server-generated informational SHA-256 digest. Clients reconcile
+ by GET rows and JSON value equality, not by reproducing this digest.
+ rowCount:
+ type: integer
+ minimum: 0
+ errors:
+ type: array
+ items:
+ $ref: '#/components/schemas/Issue'
+ derivatives:
+ $ref: '#/components/schemas/Phase'
+ File:
+ type: object
+ additionalProperties: true
+ required:
+ - name
+ - sizeBytes
+ - sha256
+ - state
+ - mediaType
+ properties:
+ name:
+ type: string
+ minLength: 1
+ description: Exact accepted multipart filename; also the row reference.
+ sizeBytes:
+ type: integer
+ minimum: 0
+ sha256:
+ type: string
+ pattern: ^[a-f0-9]{64}$
+ state:
+ type: string
+ enum:
+ - complete
+ mediaType:
+ type: string
+ minLength: 1
+ Manifest:
+ type: object
+ additionalProperties: true
+ required:
+ - submissionId
+ - revision
+ - files
+ properties:
+ submissionId:
+ type: string
+ format: uuid
+ revision:
+ type: integer
+ minimum: 0
+ files:
+ type: array
+ items:
+ $ref: '#/components/schemas/File'
+ Issue:
+ type: object
+ additionalProperties: true
+ required:
+ - code
+ - message
+ properties:
+ code:
+ type: string
+ minLength: 1
+ message:
+ type: string
+ minLength: 1
+ clientRowId:
+ type: string
+ minLength: 1
+ rowIndex:
+ type: integer
+ minimum: 0
+ field:
+ type: string
+ minLength: 1
+ limit:
+ type: number
+ Validate:
+ type: object
+ additionalProperties: false
+ properties: {}
+ description: Empty object; If-Match selects the input revision.
+ Validation:
+ type: object
+ additionalProperties: true
+ required:
+ - id
+ - submissionId
+ - revision
+ - valid
+ - configDigest
+ - manifestDigest
+ - errors
+ - warnings
+ - normalizedRows
+ - effectiveOwnerId
+ - processing
+ properties:
+ id:
+ type: string
+ format: uuid
+ submissionId:
+ type: string
+ format: uuid
+ revision:
+ type: integer
+ minimum: 0
+ valid:
+ type: boolean
+ configDigest:
+ type: string
+ minLength: 1
+ manifestDigest:
+ type: string
+ minLength: 1
+ errors:
+ type: array
+ items:
+ $ref: '#/components/schemas/Issue'
+ warnings:
+ type: array
+ items:
+ $ref: '#/components/schemas/Issue'
+ normalizedRows:
+ type: array
+ items:
+ $ref: '#/components/schemas/Row'
+ effectiveOwnerId:
+ type: string
+ format: uuid
+ processing:
+ $ref: '#/components/schemas/Processing'
+ Commit:
+ type: object
+ additionalProperties: false
+ required:
+ - validationId
+ properties:
+ validationId:
+ type: string
+ format: uuid
+ Accepted:
+ type: object
+ additionalProperties: true
+ required:
+ - submissionId
+ - operationId
+ - importTaskId
+ - acceptedRevision
+ - statusUrl
+ properties:
+ submissionId:
+ type: string
+ format: uuid
+ operationId:
+ type: string
+ format: uuid
+ importTaskId:
+ type: string
+ format: uuid
+ acceptedRevision:
+ type: integer
+ minimum: 0
+ statusUrl:
+ type: string
+ minLength: 1
+ Phase:
+ type: object
+ additionalProperties: true
+ required:
+ - state
+ properties:
+ state:
+ type: string
+ enum:
+ - pending
+ - running
+ - complete
+ - failed
+ - skipped
+ - unknown
+ description: Indexing unknown means submitted to the existing async indexing
+ queue; no completion acknowledgment is available. Derivatives unknown
+ requires operator reconciliation.
+ message:
+ type: string
+ minLength: 1
+ ResultRow:
+ type: object
+ additionalProperties: true
+ required:
+ - clientRowId
+ - encounterIds
+ - occurrenceIds
+ - individualIds
+ - mediaAssetIds
+ properties:
+ clientRowId:
+ type: string
+ minLength: 1
+ encounterIds:
+ type: array
+ items:
+ type: string
+ format: uuid
+ occurrenceIds:
+ type: array
+ items:
+ type: string
+ minLength: 1
+ individualIds:
+ type: array
+ items:
+ type: string
+ minLength: 1
+ mediaAssetIds:
+ type: array
+ items:
+ type: integer
+ minimum: 0
+ Results:
+ type: object
+ additionalProperties: true
+ required:
+ - submissionId
+ - state
+ - indexing
+ - detection
+ - identification
+ - rows
+ - errors
+ properties:
+ submissionId:
+ type: string
+ format: uuid
+ state:
+ type: string
+ enum:
+ - draft
+ - validated
+ - queued
+ - importing
+ - imported
+ - failed
+ - needs_reconciliation
+ - cancelled
+ - expired
+ indexing:
+ $ref: '#/components/schemas/Phase'
+ detection:
+ $ref: '#/components/schemas/Phase'
+ identification:
+ $ref: '#/components/schemas/Phase'
+ rows:
+ type: array
+ items:
+ $ref: '#/components/schemas/ResultRow'
+ nextCursor:
+ type: string
+ minLength: 1
+ errors:
+ type: array
+ items:
+ $ref: '#/components/schemas/Issue'
+ links:
+ type: object
+ additionalProperties:
+ type: string
+ description: Authorized task and record URL links; clients must not construct
+ page paths from IDs.
+ derivatives:
+ $ref: '#/components/schemas/Phase'
+ Capabilities:
+ type: object
+ additionalProperties: true
+ required:
+ - contractVersion
+ - admissionEnabled
+ - commitEnabled
+ - processingModes
+ - authentication
+ - limits
+ - rowFields
+ properties:
+ contractVersion:
+ type: string
+ enum:
+ - '1'
+ admissionEnabled:
+ type: boolean
+ commitEnabled:
+ type: boolean
+ processingModes:
+ type: array
+ items:
+ type: string
+ enum:
+ - import-only
+ - detect
+ - detect-and-identify
+ authentication:
+ type: array
+ items:
+ type: string
+ enum:
+ - bearer
+ limits:
+ type: object
+ additionalProperties: false
+ required:
+ - maxRows
+ - maxFileBytes
+ - maxRequestBytes
+ - maxDraftBytes
+ - maxMediaPerEncounter
+ - maxActiveJobs
+ - draftTtlSeconds
+ - idempotencyRetentionSeconds
+ properties:
+ maxRows:
+ type: integer
+ minimum: 1
+ maxFileBytes:
+ type: integer
+ minimum: 1
+ maxRequestBytes:
+ type: integer
+ minimum: 1
+ maxDraftBytes:
+ type: integer
+ minimum: 1
+ maxMediaPerEncounter:
+ type: integer
+ minimum: 1
+ maxActiveJobs:
+ type: integer
+ minimum: 1
+ draftTtlSeconds:
+ type: integer
+ minimum: 1
+ idempotencyRetentionSeconds:
+ type: integer
+ minimum: 1
+ description: Minimum guaranteed retention. The pilot retains records
+ indefinitely; no automatic database purge.
+ maxFieldsPerRow:
+ type: integer
+ enum:
+ - 256
+ maxDraftsPerUser:
+ type: integer
+ enum:
+ - 20
+ maxNewDraftsPerDay:
+ type: integer
+ enum:
+ - 20
+ description: Per owner in a rolling 24-hour window; cancellation does
+ not refund this budget.
+ rowFields:
+ type: object
+ additionalProperties: true
+ properties:
+ supported:
+ type: array
+ items:
+ type: string
+ required:
+ type: array
+ items:
+ type: string
+ indexedMedia:
+ type: string
+ stagingAvailable:
+ type: boolean
+ maxFiles:
+ type: integer
+ enum:
+ - 200
+ maxImagePixels:
+ type: integer
+ enum:
+ - 24000000
+ uploadMediaTypes:
+ type: array
+ items:
+ type: string
+ StoredRows:
+ type: object
+ properties:
+ rows:
+ type: array
+ items:
+ $ref: '#/components/schemas/Row'
+ required:
+ - rows
+ additionalProperties: true
diff --git a/docs/design/submissions/pilot-runbook.md b/docs/design/submissions/pilot-runbook.md
new file mode 100644
index 0000000000..cc7bf820db
--- /dev/null
+++ b/docs/design/submissions/pilot-runbook.md
@@ -0,0 +1,209 @@
+# Submissions pilot: operator and integration handoff
+
+Implementation is local and disabled by default. No QA or production deployment,
+partner enrollment, or live import has been performed. Verification is recorded in README.md; outstanding deployment checks below are release gates.
+
+## Installation controls
+
+In the installation's private `apiAccessKeys.properties` override (read fresh on each
+policy check, independent of browser/user configuration caches):
+
+```properties
+submissions.enabled=false
+submissions.commitEnabled=false
+submissions.workerEnabled=false
+submissions.stagingDirectory=/srv/wildbook-private/submissions
+submissions.allowedUserIds=,
+```
+
+Use existing trusted integration users, with real usernames, and the installation's
+existing RSA JWT configuration. Start with one or two partners. Context0 only.
+Setting `enabled` permits new mutations; `commitEnabled` independently permits
+queue acceptance. `workerEnabled` starts the lifecycle-managed worker at application
+startup and controls processing at runtime. Restart after enabling workers for the
+first time. Admission can be disabled while accepted jobs drain. Removing a partner
+blocks new writes and causes unstarted imports for that owner to fail eligibility.
+Owners can continue status/results reads with a submissions:read token.
+
+Create the staging directory outside the webapps tree, legacy upload directory,
+import directory, and every local asset-store root. Give only the service account access (0700 where supported).
+Use a filesystem shared by all application instances, with capacity monitoring.
+Do not share this directory across contexts or separate installations. This pilot
+only supports context0. The new table has not been deployed; do not reuse an
+experimental SUBMISSION table with an older schema without an explicit migration.
+The default media-size configuration still bounds each file. Configure Tomcat's
+upload/read timeout (for example `disableUploadTimeout="false"` and an appropriate
+`connectionUploadTimeout`) and reverse-proxy request deadlines: a blocked idle
+socket cannot be interrupted by the application's per-read two-minute deadline.
+The application admits two expensive intake operations per JVM, at most one per
+owner. A worker uses one slot, leaving a slot for uploads/validation. Contention returns
+429; this conservative setting intentionally limits pilot throughput.
+
+Build with DataNucleus enhancement. Apply the additive SUBMISSION table metadata
+from `src/main/resources/org/ecocean/submission/package.jdo` using the installation's
+normal schema rollout process, including the create-key uniqueness constraint and
+LOCK_VERSION. When upgrading a development schema from an earlier increment, backfill new
+primitive timestamp columns with zero before enforcing NOT NULL. Job/report/result
+strings may be null on older drafts. Test schema
+creation and upgrades in isolated PostgreSQL before deployment. Do not point a
+local build or test run at a production database. The existing biological tables
+and bulk-import defaults do not need a data migration.
+
+The existing Maven WAR configuration excludes `WEB-INF/web.xml`. Deploy the new
+servlet and Shiro filter mappings from `src/main/webapp/WEB-INF/web.xml` through
+the installation's descriptor rollout process as well as deploying the WAR.
+Verify the running descriptor contains both submissions route patterns and their
+Bearer filter before enabling admission; a WAR-only update is insufficient.
+
+## Agent skill
+
+After deployment, the public toolbox at `/api/v3/agent-skill` links to
+`/api/v3/agent-skill/submit-sightings`. Give that skill URL to a coding agent along
+with the installation URL and its separately supplied scoped token. The skill
+contains the exact supported row fields, settings discovery, request formats,
+field-specific validation failures and safe recovery instructions. Public skill
+availability does not enable intake or enroll an account. Include fetching the
+base toolbox and new skill in QA deployment checks.
+
+## Authentication and reference client
+
+Mint a bearer token using fresh HTTP Basic credentials at
+`POST /api/v3/auth/token?scope=submissions:write`. Do not put credentials in a URL.
+The response contains `token`, `tokenType`, `expiresInSeconds`, and `scope`.
+Renewal requires fresh credentials. A read token uses `scope=submissions:read`.
+Omitting scope retains legacy identity-only issuance. Scoped submission tokens have
+a separate JWT audience and cannot be used with legacy search routes. Browser
+cookie authentication is not accepted on the submissions resource.
+
+Store a token in `WILDBOOK_SUBMISSIONS_TOKEN`, then use the POSIX Python 3 client:
+
+```bash
+python3 scripts/submissions/client.py \
+ --base-url https://your-qa-installation.example \
+ --rows rows.json --media-dir ./photos --state ./submission-state.json
+```
+
+This creates/uploads/validates without committing. Inspect the reported validation
+errors, correct rows.json and rerun with the same state file to update the same
+draft. The client checks that the server still holds its previously saved rows
+before replacing them. Repeat with `--commit` when ready. Keep the same state file for all
+retries and restarts. It contains IDs and operation keys, not credentials. A file
+lock prevents concurrent clients using that state on a local POSIX filesystem;
+network filesystem and cross-host locking are not supported. Use one state file per batch.
+The client refuses redirects and remote cleartext HTTP, disables environment
+proxies, and honors Retry-After with up to five minutes of contention retries. Renew expired tokens and
+resume with the same state. Never create a replacement batch merely because a
+commit timed out. Use `--cancel --base-url ... --state ...` to cancel an editable
+draft without supplying rows/media. If the create response was lost before its ID
+was saved, first rerun the normal command to recover the ID using its saved key.
+A frozen commit that was never accepted can
+be cleared using `--reset-commit`: the client first requires server state draft
+or validated with no operation ID. Then correct/revalidate and commit. Accepted,
+failed or uncertain executions cannot be reset or cancelled through this client.
+Changed image content needs a new filename, or cancel the draft and start a new
+state file; completed file contents are immutable.
+
+Example `rows.json` for one new encounter with two photographs:
+
+```json
+{"rows":[{"clientRowId":"camera-observation-001","fields":{
+ "Encounter.genus":"Manta",
+ "Encounter.specificEpithet":"birostris",
+ "Encounter.year":2026,
+ "Encounter.locationID":"REPLACE_WITH_CONFIGURED_LOCATION",
+ "Encounter.mediaAsset0":"photo-1.jpg",
+ "Encounter.mediaAsset1":"photo-2.jpg"
+}}]}
+```
+
+Use the installation's configured taxonomy and location. Discover exact supported
+fields, modes and limits at GET `/api/v3/submissions/capabilities`. The pilot creates
+one encounter per row; explicit encounter, individual, sighting, project and owner
+fields are unavailable. It accepts JPEG and PNG with matching filename extensions.
+Each photograph belongs to one row; duplicate references are rejected. Optional
+date parts remain optional; a year-only observation stays year-only. Limits are
+200 rows, 256 fields per row, 200 files and 200 MiB completed bytes per draft, 20
+live drafts, 20 new drafts per rolling 24 hours, and one active job per owner.
+Cancellation does not refund the daily creation budget. File-count and byte overages return 413.
+
+Use GET rows/files to reconcile lost edit/upload responses. PUT rows replaces the
+whole row list. Edits require the current quoted If-Match revision and invalidate
+validation. Validate returns 200 with `valid=false` for data errors; it does not
+advance the revision. Commit requires a current validation ID, revision and saved
+Idempotency-Key. Same-key/same-input retries recover the accepted operation;
+different keys cannot launch a second execution for the same submission.
+
+## Status, recovery and retention
+
+Poll the submission and its paginated results. Imported means domain records,
+source-row mappings and the post-import intent committed together. Detection and
+identification are skipped in this pilot. Derivative state is reported separately.
+Indexing `unknown` means submitted to the existing asynchronous indexing queue;
+this implementation does not assert search completion. Failed index dispatch
+is held as failed for operator inspection; derivative unknown leaves indexing
+pending until an operator reconciles the derivative work. Pending post-import work is scanned separately from history. Index intent is
+replayed idempotently after worker restart in five-item batches, using a startup
+watermark and bounded pagination. Check record pages/search during QA acceptance.
+
+One installation-wide import writer is claimed through PostgreSQL. A crash before
+claim leaves a queued job discoverable. A stale importing claim is moved to
+`needs_reconciliation` after an hour only when its transaction lock can be acquired.
+It is never automatically rerun. A delayed worker must recheck state under that
+same lock. Imported state wins when reconciling a lost database-commit acknowledgment.
+Interrupted derivative generation uses its own claim timestamp and is held as
+`unknown`, preserving imported records. A crashed import can pause the installation
+queue for up to one hour before the conservative reconciliation check.
+
+For uncertain work, disable commit admission and inspect the submission, reserved
+ImportTask ID, row mappings and database records using a fresh connection. Stop all
+workers before an operator repairs state. Establish whether the transaction committed
+before considering a retry. There is deliberately no public retry/requeue endpoint.
+A reconciliation patch must preserve the original operation IDs and audit the
+operator's evidence; do not change needs_reconciliation back to queued blindly.
+When resolving to imported or certainly failed, record the completion time through
+the entity transition (including `completedAt`) so staging retention can finish.
+Uncertain owners remain at their one-job limit until reconciliation.
+
+The hourly private-staging sweeper releases manifest references for expired and
+cancelled drafts under short locks. It also releases staging for imported or
+certainly failed submissions seven days after completion; imported asset-store
+originals and row mappings remain intact. Active and uncertain work is retained.
+Released manifests become empty without changing the frozen execution revision.
+Subsequent scans skip them, using keyset pages so reference removal cannot skip
+other drafts. A ten-second inventory deadline aborts physical deletion safely but
+keeps completed reference-release progress for the next pass. At most 5,000 old
+unreferenced blob directories are removed per pass. Errors are isolated from intake.
+
+Tombstones and operation keys remain in the database for retry safety; no automatic
+database purge is implemented. Monitor retained bytes and database growth before
+broader enrollment. Asset-store copies from rolled-back imports remain under the
+reserved task namespace for operator inspection. The staging sweeper never deletes
+shared or pre-existing media assets. Keep staging configuration stable while work
+is active; drain before relocating it.
+
+Monitor worker attempt/error logs by submission ID, queue age via createdAt/state,
+phase failures, 429s and staging bytes. The new deferred importer suppresses its
+raw row diagnostics. Do not log tokens or submission bodies. Application metrics
+export/dashboard integration remains an operational follow-up.
+
+## QA release gate and rollback
+
+Before enabling a partner, deploy to QA and verify:
+
+- Two-image valid import produces the expected encounter/media/task and source-row
+ mapping; partial dates retain precision and configured ownership is correct.
+- Bad location, unknown field, absent/corrupt image, stale revision and cross-owner
+ access produce the documented errors without biological records or IA work.
+- Restart after draft creation, after queue acceptance and after domain commit;
+ reconcile saved IDs and verify no duplicate execution.
+- Compare the existing browser upload/review/import/task flow against its baseline,
+ including a legacy import with grouped rows. Check derivatives and search.
+- Validate schema constraints, private storage permissions, request deadlines and
+ shared staging across instances. Toggle admission, commit and worker flags on
+ an API-only instance and verify they are read fresh. Start with one selected partner.
+
+Rollback: disable admission and commit, confirm no importing or derivative-running rows
+remain (or reconcile them), then stop
+workers before reverting application code. Retain the added table and private files
+while work or retention obligations remain. Turning off a flag alone does not undo
+an import or prove an in-flight transaction stopped.
diff --git a/docs/design/submissions/reviews/pr-handoff-review.md b/docs/design/submissions/reviews/pr-handoff-review.md
new file mode 100644
index 0000000000..b395808ac5
--- /dev/null
+++ b/docs/design/submissions/reviews/pr-handoff-review.md
@@ -0,0 +1,25 @@
+**I approve the handoff. I found no blockers.** I only read files with Read/Glob/Grep. I ran nothing and didn't repeat the code review, so every test result quoted below is from your records, not from me.
+
+## What I checked
+
+- **Scope and base:** The PR body, README (`docs/design/submissions/README.md:136-143`) and runbook all say the same thing about the base. Testing was done on `24cc99aede`, the PR targets `dcf6f460de`, and the difference between them is frontend-only. They also say the frontend tests and the WAR were not rebuilt on the new base, and that QA/browser checks on the final branch are still a release gate. I took the "Java sources and dependencies are identical" claim from your statement; I didn't check it with git.
+- **Test qualifications are honest:**
+ - The 16 failing frontend suites are explicitly *not* called pre-existing, because no base-commit comparison was run.
+ - The full Java run (1,109 tests) is placed before the skill was added. The skill-only run (15) and the `-DskipTests package` build are listed separately, and the README says tests weren't rerun during packaging.
+ - The README says `check_contract.py` "is not a complete OpenAPI conformance validator".
+ - Nothing says Claude ran anything. The final review rounds for stages 1–7 all state they were read-only with no Critical/Major findings, which matches the PR body.
+- **Deployment/readiness claims:** None are unsupported. The PR body, README, brief (`2026-09-23-submissions-engineer-brief.md:7-8`) and runbook (`pilot-runbook.md:3-4`) all say nothing is deployed, enrolled or imported. The runbook's QA gate and the `web.xml` descriptor caveat are clear. The public toolbox wording marks the skill as an enrolled pilot, and the runbook says publishing the skill doesn't enable intake.
+- **Disclosure of existing-production weaknesses:** I searched the reviews, dispositions, design docs and the skill. Every security finding is about the new, undeployed submissions path, and each has a recorded fix. The notes about existing code describe correctness or design, not anything exploitable:
+ - the legacy importer commits inside its own helpers (stage 4);
+ - the importer silently skips a missing asset;
+ - the webapp and data directory are served statically, noted only to justify where staging must go;
+ - an old JSP link was stale.
+
+ None of this is an attack path against current production. The skill contains no internal settings, allowlist details or ways around authentication.
+
+## Minor suggestions (optional)
+
+1. **Relative links in the PR body will probably break.** GitHub resolves relative paths in a PR description against the PR page, not the repo. That affects `docs/design/submissions/README.md` and the three links on line 21. Use full `.../blob/feat/submissions-api-pilot/...` URLs instead.
+2. **Stale wording in `README.md:23`:** "Review findings and dispositions will be recorded here." They're now recorded under `reviews/`, so change it to past tense.
+3. **`README.md:35` "Baseline on checkout `24cc99aede`":** consider adding "(implementation checkout; PR base is `dcf6f460de`, see below)" so the reader isn't confused before reaching line 136. The brief's "Based on … checkout `24cc99aede`" (line 4) could use the same note.
+4. **PR body line 12:** "build/WAR succeeded before the final agent-skill addition" is accurate. Adding "on `24cc99aede`" there would make each line stand on its own; line 16 already says this for the whole section.
diff --git a/docs/design/submissions/reviews/stage-1-disposition.md b/docs/design/submissions/reviews/stage-1-disposition.md
new file mode 100644
index 0000000000..b3cdae85d4
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-1-disposition.md
@@ -0,0 +1,38 @@
+# Stage 1 review disposition
+
+Claude round 1: six Major findings, no Critical findings. The user authorized
+read-only external Claude review of relevant project files.
+
+- M1: validation never bumps input revision; empty body plus If-Match; returns
+ ETag; repeat validation safely recovers a lost response.
+- M2: owner GET exposes terminal tombstones; repeated cancellation returns 204;
+ all post-acceptance/failed/uncertain states are explicitly non-cancellable.
+- M3: published stable error codes and optional issues; invalid versus stale
+ validation has distinct status/code behavior.
+- M4: added GET rows plus rowCount/rowsDigest for recovery after lost PUT response.
+- M5: occurrence/individual result IDs are nonempty strings, not forced UUIDs.
+- M6: servlet-level legacy unknown-field-default test, importer-level repeated-ID
+ grouping/year precision test, exact day diagnostic, short-array null padding.
+ Existing BulkApiPostTest duplicate/synonym tests, BulkImagesTest missing/corrupt
+ image tests and BulkApiOtherTest status/authorization fixtures cover those
+ legacy cases. Media-per-encounter enforcement is new behavior to test in stage 3;
+ no claim that these unit tests establish durable background/IA correctness.
+
+Minor fixes: clarify omission of whole processing object; remove irrelevant
+precondition responses; define exact upload filename/retry behavior; require GET
+after create replay; advertise Bearer only; permit additive response fields;
+remove YAML aliases; strengthen format/negative/example/HTTP checks; add response
+examples and result links.
+
+Round 2 and updated test results pending. Earlier baseline: 106 existing tests
+passed. Earlier new characterization run: four tests passed. These do not replace
+running the strengthened suites.
+
+Round 2: **No Major findings remain; nothing blocking.** All six Majors resolved.
+Corrected the accidental shared filename description, explicitly registered format
+checks with negative UUID/date examples, clarified {} validation input, retained
+rows during tombstone retention, marked rowsDigest informational, documented error
+mappings, and made warning/grouping assertions more precise. The required mode's
+default is documentation of whole-processing omission, not schema default filling.
+Strengthened suites passed 21 tests; final small assertion changes are rerun with
+the next test gate. Stage 1 is cleared for runtime implementation.
diff --git a/docs/design/submissions/reviews/stage-1-round-1.md b/docs/design/submissions/reviews/stage-1-round-1.md
new file mode 100644
index 0000000000..adc634966d
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-1-round-1.md
@@ -0,0 +1,59 @@
+I found no Critical findings and 6 Major ones. The biggest gaps are in the retry and revision rules, and in the error shape a client needs to decide what to do after a failure. I didn't run anything, including `check_contract.py` and the new test, as you asked.
+
+## Major
+
+**M1. Whether validate changes the revision is undefined.** (`openapi.yaml:604-618`, `1097-1103`; `examples.json:62,71`)
+- The example validation is at revision 2, but the accepted commit shows `acceptedRevision: 3`. That suggests the draft → validated change bumps the revision.
+- The validate 200 response returns no `ETag`, though, so the client doesn't know which `If-Match` value to send on commit.
+- It's also unclear what happens if the `If-Match` header and the body `revision` disagree.
+- There's no way to fetch a validation by ID, so a client that loses the validate response can't recover it except by validating again.
+- **Fix:** Pick one of two rules.
+ - Validate never bumps the revision, and "validated" is derived from `validationId` matching the current revision. Then re-validating the same revision is safe.
+ - Or validate returns an `ETag` plus a `resultingRevision`, and `GET /{id}/validations/{validationId}` is added.
+
+ Either way, drop the body `revision` or state that it must equal `If-Match` (otherwise 400).
+
+**M2. Cancellation, 410 and the terminal states contradict each other.** (`openapi.yaml:231-236`, `239-241`, `289-294`, `1025-1034`)
+- DELETE promises 204 on repeated cancellation but also lists 410 for cancelled drafts.
+- GET returns 410 for cancelled/expired drafts, so the `cancelled` and `expired` states in `Submission` can never be returned.
+- "Replay of the cancellation revision" doesn't say which `If-Match` value a retry must carry: the revision before cancellation, or the one after.
+- **Fix:** For the owner, a repeated DELETE returns 204 whatever `If-Match` says (skip the precondition check once the draft is already cancelled). Remove 410 from DELETE. Then either remove `cancelled`/`expired` from the `Submission` enum, or have GET return 200 with that state and keep 410 only for results/files. Also say whether a failed or `needs_reconciliation` submission can be cancelled.
+
+**M3. The error body can't be acted on by a program.** (`openapi.yaml:916-926`)
+- The plan (line 75) requires optional row/field/limit details, but `Error` has only `code`, `message` and `requestId`, and it forbids extra properties.
+- One 409 covers "State, key or content conflict", and no codes are listed. A client can't tell a reused idempotency key from "already committed" or "commit blocked because validation is invalid".
+- 413 and 429 can't report which limit was hit.
+- **Fix:** Add `issues: Issue[]` (optional) to `Error` and a published enum or table of stable `code` values. At minimum: `PRECONDITION_REQUIRED`, `REVISION_STALE`, `IDEMPOTENCY_KEY_REUSED`, `ALREADY_COMMITTED`, `VALIDATION_STALE`, `VALIDATION_INVALID`, `INVALID_STATE`, `LIMIT_EXCEEDED`, `ADMISSION_DISABLED`, `DUPLICATE_CLIENT_ROW_ID`. Also say which status commit returns when `valid=false`.
+
+**M4. A lost response to a rows PUT can't be recovered.** (`openapi.yaml:325-329`, `1002-1048`)
+- After a lost response, the retry gets 412 because the revision has moved on.
+- `Submission` exposes neither the rows nor a digest of them, so the client can't tell whether its write was applied. Uploads have a reconcile path; rows don't.
+- **Fix:** Add `rowsDigest` (SHA-256 of the canonical rows) and `rowCount` to `Submission`, or add `GET /{id}/rows`. Then document the recovery rule: if the digest matches, treat the write as successful.
+
+**M5. `occurrenceIds` is declared as UUIDs, but real occurrence IDs aren't always UUIDs.** (`openapi.yaml:1194-1196`)
+- `BulkImporter.getOrCreateOccurrence` (`BulkImporter.java:1036-1042`) uses whatever string the user put in `Sighting.sightingID`/`Encounter.sightingID`.
+- Existing occurrences can have non-UUID IDs, so a valid response would fail the contract. `individualIds` carries a similar risk for legacy data.
+- **Fix:** Make `occurrenceIds` (and probably `individualIds`) plain `type: string, minLength: 1`.
+
+**M6. The characterization tests don't pin the behaviour the new API depends on.** (`BulkSubmissionCompatibilityTest.java`)
+- **Unknown-field default (`:52-59`):** the test only exercises `BulkValidatorException.treatAsWarning(true/false)`. It doesn't pin the actual legacy default, `badFieldnamesAreWarnings=true` (`BulkImport.java:183-184`), which is the behaviour the new API deliberately differs from. Either extract that default into a constant or helper and assert it, or cover it in a `BulkApiPostTest`-style servlet test.
+- **Year precision (`:29-30`):** the checks that `Encounter.month` and `Encounter.day` are absent pass trivially because neither was in the input. That proves nothing about precision. Assert at importer level that a year-only row produces an encounter with no month/day, or rename the test.
+- **Feb 30 (`:64`):** `anyMatch` would pass on any unrelated error. Assert `result.get("Encounter.day") instanceof BulkValidatorException`, plus the message or code.
+- **Object vs array rows (`:35-50`):** this skips the one real difference, which is that a short array pads with nulls (`BulkImportUtil.java:38-42`). Add a short-array case.
+- **Coverage vs the exit gate:** the plan (lines 87-90) lists synonyms, row grouping and media count, missing/corrupt images, and status response shapes. None are covered here or cited from the existing suites. Grouping matters most, because `maxMediaPerEncounter` depends on it (plan line 174). Add them, or state which existing test covers each one.
+
+## Minor
+
+1. **Default processing mode (`openapi.yaml:944-953` vs `74-75`):** `Processing.required: [mode]` means `default: import-only` never applies. The "omitted means import-only" rule exists only in prose, and `check_contract.py:57` asserts a default that can't take effect. Either make `mode` optional, or document that `processing` is omitted as a whole.
+2. **Irrelevant error statuses (`openapi.yaml:141-164`, and 404 on create/capabilities):** create has no `If-Match`, so 412/428 can't happen, and 404 doesn't apply there. Listing them misleads generated clients. Only list 412/428 where `If-Match` is required.
+3. **Upload name mapping (`openapi.yaml:493-498`, `1059`):** it isn't stated that `File.name` is exactly the multipart filename that rows reference in `Encounter.mediaAssetN`. Nor is it stated that names are rejected rather than silently cleaned (plan lines 164-165). Also say whether a same-content retry bumps the revision.
+4. **Create replay (`openapi.yaml:74`):** a replay returns the original 201 body and ETag, which may be out of date. Tell clients to GET before mutating.
+5. **Session auth (`openapi.yaml:882-888` vs `1262-1268`):** capabilities can advertise `session`, but no cookie security scheme or CSRF header is declared.
+6. **Strict responses:** `additionalProperties: false` on response schemas blocks adding fields later. Keep it on request schemas and loosen it on responses.
+7. **YAML anchors:** schemas are shared via anchors (`&id001` is defined in a query parameter at `:817` and reused throughout components). This is fragile to edit. Replace with named `$ref` schemas (`NonEmptyString`, `Uuid`, `Revision`).
+8. **`check_contract.py` checks less than it claims:**
+ - It validates without a format checker (`:48,53`), so `uuid` and `date-time` are never enforced. Use `format_checker=Draft4Validator.FORMAT_CHECKER` or an equivalent.
+ - There are no negative examples, such as a null field or an unknown property being rejected.
+ - It doesn't assert 412/428 wherever `If-Match` is required, `Retry-After` on 429, `ETag` on GET/mutation 2xx responses, or `Idempotency-Key` on create.
+9. **Examples are incomplete (`examples.json`):** the plan (lines 54-55) calls for examples across the full create → upload → rows → validate → commit → results sequence, including errors. Missing are `Manifest`, `Capabilities`, a valid `Validation`, and `Error` bodies for 409, 412, 428 and 429.
+10. **Results links:** the plan (line 242) asks for links to the existing task and record pages, but `Results` has no field for them.
diff --git a/docs/design/submissions/reviews/stage-1-round-2.md b/docs/design/submissions/reviews/stage-1-round-2.md
new file mode 100644
index 0000000000..3038143dc1
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-1-round-2.md
@@ -0,0 +1,35 @@
+**No Major findings remain.** All six round-1 Majors are fixed in the contract and tests. There are no Critical findings and nothing blocking. I found one new wording defect worth fixing before the contract is published (item 1). Everything else is Minor. I didn't run anything, as you asked. The test findings below come from reading the code, so the strengthened test run you have in progress is still the real check.
+
+## Round-1 Majors
+
+| # | Status | Evidence |
+|---|---|---|
+| M1 | Resolved | `openapi.yaml:692-711`: validate never bumps the revision, returns an `ETag`, is safe to repeat, and has no body `revision`. The example revisions agree: manifest 1 → validation 2 → `acceptedRevision` 2. |
+| M2 | Resolved | `:165-167`: GET returns 200 with a tombstone. `:232-235`: a repeat DELETE returns 204 whatever `If-Match` says, and the other states return 409. 410 is no longer listed on DELETE. |
+| M3 | Resolved | `:1033-1073`: stable `code` enum plus optional `issues`. `:807-809`: invalid, stale and already-committed validations now have distinct status/code pairs. |
+| M4 | Resolved | `GET /rows` added, plus `rowCount` (required) and `rowsDigest`. |
+| M5 | Resolved | `:1418-1427`: `occurrenceIds` and `individualIds` are now non-empty strings. |
+| M6 | Resolved | Unknown-field default is pinned at servlet level: `BulkApiPostTest.java:464-488` uses no `tolerance` override and checks warnings=1, errors=0. Grouping and year-only precision are tested at importer level (`:490-524`). The Feb 30 check is exact (`Compat:64-66`). The short-array null padding case is added (`:69-79`). The disposition says which existing suites cover the remaining cases. |
+
+## Findings
+
+1. **Medium (fix before publishing), `openapi.yaml`: one description was pasted onto unrelated fields.** The text "Exact accepted multipart filename; also the row reference." now appears on the results `cursor` (`:929`), `Error.message`/`requestId` (`:1065,1069`), all four `Issue` string fields (`:1270-1285`), `configDigest`/`manifestDigest` (`:1323,1327`), `File.mediaType` (`:1241`), `statusUrl` (`:1379`), `Phase.message` (`:1398`), `ResultRow.clientRowId` (`:1412`) and `nextCursor` (`:1473`). Generated clients and docs would describe these fields wrongly. It looks like the YAML-alias removal went wrong. **Fix:** keep that description only on `File.name` (`:1227`). Remove it everywhere else, or replace it with a correct one-line description.
+
+2. **Minor, `check_contract.py:55-57`: the format check probably isn't doing anything.** From memory of jsonschema's `_format.py` (please confirm against the installed version), `uuid` is only registered for Draft 2019-09 and 2020-12, not Draft 4. `date-time` is only checked if `rfc3339-validator` is installed; otherwise it's skipped without warning. So the Minor 8 fix may have no effect. **Fix:** add a negative example with `"id": "not-a-uuid"` (and a bad `date-time`) marked `valid: false`, so the checker fails if formats aren't enforced. Or register those checkers on the `FormatChecker` explicitly.
+
+3. **Minor, `openapi.yaml:796-801` vs `stage-1-disposition.md:6`: validate body.** The disposition says "empty body", but `requestBody.required: true` with the `Validate` schema means clients must send `{}`. **Fix:** either say "send `{}`" in the description, or set `required: false`.
+
+4. **Minor, `openapi.yaml:430-496`: `GET /rows` has no 410.** GET `/files` (`:556`) returns 410 for tombstones, but GET `/rows` doesn't list it. **Fix:** add 410 to `GET /rows`, or state in the description that rows stay readable during tombstone retention.
+
+5. **Minor, `openapi.yaml:432-435` vs `:1208-1210`: row comparison rules don't match.** GET `/rows` promises the "exact stored rows" but tells clients to compare "normalized JSON values". Meanwhile `rowsDigest` canonicalization doesn't say how numbers are written (`2026` vs `2026.0`). **Fix:** say that GET returns the rows as accepted and that clients compare by JSON value equality. Either state that `rowsDigest` uses RFC 8785 (JCS) canonical JSON, or mark it informational only.
+
+6. **Minor, `openapi.yaml`: some error mappings are still missing or irrelevant.**
+ - DELETE still lists 413 and 422 (`:295-306`), which can't happen. Remove them.
+ - Validate lists 422, but its description says errors return 200 with `valid=false`. Say when 422 applies (e.g. no rows yet), or remove it.
+ - Two status mappings are unstated. Rows PUT with a duplicate `clientRowId` (409 or 422?) and DELETE in a non-cancellable state (409 `INVALID_STATE`?). Add a one-line code→status note or table under `Error`.
+
+7. **Minor, `openapi.yaml:1100` and `check_contract.py:61-62`: leftover processing default.** `default: import-only` on the required `mode` field can never take effect. The prose now explains the rule correctly, but the checker still asserts the default. **Fix:** remove the `default` and the assertion, or leave both as documentation only. This is cosmetic.
+
+8. **Minor, `BulkApiPostTest.java:485-486`: the warning test doesn't check where the warning came from.** Any single warning makes it pass. **Fix:** assert that the one warning has `fieldName == "Unknown.field"` and the unknown-fieldname type. With `verbose` set, the response may include it; if not, capture it from `dataWarnings` via the verbose output.
+
+9. **Minor, `BulkApiPostTest.java:519`: the grouping test depends on `.get(0)`.** It assumes the Shepherd built first is the one that stores encounters. **Fix:** loop over `sh.constructed()` and assert `storeNewEncounter` was called exactly once in total, so the test doesn't depend on construction order.
diff --git a/docs/design/submissions/reviews/stage-2-disposition.md b/docs/design/submissions/reviews/stage-2-disposition.md
new file mode 100644
index 0000000000..bb4ed61b36
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-2-disposition.md
@@ -0,0 +1,31 @@
+# Stage 2 review disposition
+
+Round 1: one Major, no Critical findings.
+
+- Major: submission capabilities now use the configured JWT audience plus
+ `/submissions`; identity verification keeps its original audience. The legacy
+ token filter also rejects submission-scope claims. Added a real-signature test
+ proving the legacy search filter returns 401 for a submission token.
+- Added maxDraftsPerUser/maxFieldsPerRow to discovery and published schemas.
+- Noneditable PUT rows consistently returns 409; If-Match length matches code.
+- Draft-capacity 429 no longer suggests a five-second retry.
+- Authentication completes before chain.doFilter, preserving downstream errors.
+- Added bounded strict JSON parsing, explicit UTF-8 validation, duplicate-key and
+ trailing-input rejection, and parser tests. Jackson core 2.17.0 was already a
+ transitive runtime dependency; it is now an explicit pinned compile dependency.
+- Added real PostgreSQL quota-race, expiry and admin-access tests.
+- Kept admission checks ahead of routing for writes: disabled or unenrolled callers
+ do not reach the new mutating resource dispatch. This cosmetic error precedence
+ is intentional; capabilities remains readable.
+
+Local database testing also caught external writes to public persistent fields
+not marking objects dirty. Fields are now private and mutations are performed
+inside the enhanced entity's methods. Durability, competing revision and confirmed
+rollback tests passed after that correction. A mock header-count assertion was
+fixed to allow setting the same no-store header more than once.
+
+Round 2: Claude found no Critical or Major issues. The expanded test run passed
+38 tests with zero failures, errors or skips. Remaining Unicode input handling
+and published schema bounds were corrected; unsupported-method 405 documentation
+is tracked for the final contract publication. No upload,
+validation, commit or worker implementation is claimed in this stage.
diff --git a/docs/design/submissions/reviews/stage-2-round-1.md b/docs/design/submissions/reviews/stage-2-round-1.md
new file mode 100644
index 0000000000..9af9ad71ed
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-2-round-1.md
@@ -0,0 +1,59 @@
+No Critical findings. I have one Major finding, and it's about scope containment, not privilege escalation. Everything else is Minor. I only read the code; I didn't run any tests.
+
+## Major
+
+**1. Submission-scoped tokens also work anywhere identity tokens work.**
+`JwtService.signSubmission` (`api/auth/JwtService.java:104`) uses the same issuer and audience as identity-only tokens. It only adds a `submissionScope` claim.
+- `WildbookTokenAuthenticationFilter` (`security/WildbookTokenAuthenticationFilter.java:73-76`) never looks at that claim. So a `submissions:read` or `submissions:write` token is accepted on `/api/v3/search/**` and `/api/v3/media/resolve`.
+- The external scoped-access service described in the JwtService Javadoc would also accept it as a full identity token.
+- Blocking unscoped tokens from submissions was done. Keeping scoped tokens out of everything else was not.
+- The practical risk is limited: minting needs the same password as an identity token, so no one gains privileges. But a leaked pilot client token gives search access for the token's lifetime (up to 24h), not just access to drafts. Since scoped credentials are the point of this stage, I'd fix it now.
+- **Fix:** mint submission tokens with a separate audience, e.g. a new `jwtSubmissionAudience` setting defaulting to `wildbook-submissions`. Add a `verify(token, audience)` overload and use it in `SubmissionAuthenticationFilter`. That keeps both the search filter and the external service rejecting these tokens with no change on their side. As an extra safeguard, have `WildbookTokenAuthenticationFilter` reject any token that has a `submissionScope` claim. Add a test showing a scoped token gets 401 on search.
+
+## Minor
+
+2. **The published spec and the capabilities response disagree.** `SubmissionApiCapabilities.limits` has `additionalProperties: false` (`openapi.yaml:396`), but `Submissions.java:76` returns `maxDraftsPerUser`. Strict generated clients will reject the response. Also, `stage-2-operations.md:32-33` says the 256-fields-per-row limit is published in capabilities, but it isn't. Fix: add `maxDraftsPerUser` and `maxFieldsPerRow` to the schema, and return `maxFieldsPerRow`.
+
+3. **PUT rows documents 410 for expired or cancelled drafts, but the code returns 409.** `openapi.yaml:2084` lists 410 for "Expired or cancelled draft". `SubmissionStore.editable` always returns 409 `INVALID_STATE`, which matches what DELETE documents. Fix: remove the 410 from PUT, or return 410 `GONE` there.
+
+4. **The 20-draft quota says to retry in 5 seconds.** `SubmissionAuthenticationFilter.error` sends `Retry-After: 5` on every 429. The draft cap won't clear in 5 seconds; it clears when a draft is cancelled or expires. Clients that honour the header will keep retrying pointlessly. Fix: leave out `Retry-After` for this quota (or report when the oldest draft expires), and keep it for real rate limits.
+
+5. **The filter's `try` block also wraps `chain.doFilter`** (`SubmissionAuthenticationFilter.java:71-76`). Any `IOException`, `ServletException` or runtime exception from later in the chain is logged as a generic 503 "authentication unavailable". It may also be written onto a response that has already been committed. Fix: finish authentication inside the `try`, then call `chain.doFilter` outside it.
+
+6. **Request bodies are parsed leniently.** `new JSONObject(String)` in org.json 20240303 accepts unquoted keys and values, single quotes, and (I believe) extra text after the closing `}`. `Submissions.body` also silently replaces invalid UTF-8 bytes, which changes stored field values. The strict key and type checks still bound the envelope, so the risk is low. Fix: decode with a `CharsetDecoder` set to `REPORT`, and reject trailing content after parsing (or parse strictly with Jackson). Only enrolled writers can reach the parser, so the unbounded nesting depth risk is small, but a depth-capped parser would also cover it.
+
+7. **Error order: admission is checked before routing** (`Submissions.java:23`). A PUT or POST to a route that doesn't exist returns 503 or 403 instead of 404. That's cosmetic, but it misleads clients probing capabilities.
+
+8. **The If-Match pattern differs slightly.** The spec's `^"[0-9]+"$` accepts values that the code rejects with 400 (`{1,18}` digits). Add `maxLength: 20` or a bounded pattern to the spec.
+
+9. **Some behaviours this stage relies on aren't tested:**
+ - Concurrent creates with *distinct* keys near the 20-draft cap. That case is the reason for the per-owner lock.
+ - Logical expiry: an expired draft stays readable, rejects edits with 409, and doesn't count toward the quota.
+ - An admin accessing another owner's draft.
+ - The negative test for finding 1.
+
+ The existing competing-edit test is good. Getting 412 rather than 503 shows the advisory lock is serializing, not just the JDO version check.
+
+## Checked and correct
+- Transaction handling:
+ - No optimistic-transaction setting in `jdoconfig`, so JDO runs datastore transactions and `pg_advisory_xact_lock` is held on the transaction's own connection until commit or rollback.
+ - The `JDOConnection` is closed before further persistence-manager calls.
+ - Lookups use `setIgnoreCache` plus `refresh`.
+ - Responses are built before commit.
+ - `commitDBTransactionWithStatus` maps a failed commit to 503.
+- Create idempotency:
+ - The key hash is scoped to context, owner and operation.
+ - Canonical hashing makes an omitted `processing` equal to an explicit import-only one.
+ - The replay check runs before the quota check.
+ - A replay returns exactly the original body, as the test confirms.
+- Ownership: non-owners get 404, the context is checked, and admin status comes from the token's user.
+- Session handling: the filter never calls `login` or `getSession`, and `isUserInRole` ignores roles from the cookie session.
+- Gates:
+ - GET works with either scope and has no admission check.
+ - Writes need write scope and re-check enrollment in both the filter and the servlet.
+ - AuthToken only grants write scope to enrolled users while admission is enabled.
+- Existing tokens: `sign()` without a scope produces the same claims as before, and the Shiro `[urls]` order puts `submissionAuth` ahead of the bulk rules. Bulk and browser routes are unchanged.
+- Cancellation: a repeat cancel returns 204 before the revision comparison, as documented, and editing an expired draft returns 409.
+- The JDO mapping (UUID primary key, unique key hash, `LOCK_VERSION`, `LONGVARCHAR`) matches the design. The enhancer's default includes pick up `package.jdo`.
+
+Once finding 1 is fixed, I'd consider stage 2 converged with no Major findings. The Minor items can go into this change or be tracked separately.
diff --git a/docs/design/submissions/reviews/stage-2-round-2.md b/docs/design/submissions/reviews/stage-2-round-2.md
new file mode 100644
index 0000000000..04133f2762
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-2-round-2.md
@@ -0,0 +1,29 @@
+**Stage 2 has converged: I found no Critical or Major issues.** This was a read-only review. I didn't run anything, so the expanded PostgreSQL test run you have going is still the evidence the tests actually pass.
+
+## The round 1 Major is fixed
+- **Tokens are now separated by audience.** Submission tokens are minted with the configured audience plus `/submissions` (`api/auth/JwtService.java:117`). Identity checks require the original audience (`JwtService.java:128-130`) and submission checks require the suffixed one (`:132-134`). The token library matches the audience exactly, so each kind of token fails the other's check. The same applies to the external service, provided it also checks the audience.
+- **Both filters use the right check.** `security/SubmissionAuthenticationFilter.java:50` uses `verifySubmission`. `security/WildbookTokenAuthenticationFilter.java:75-78` also rejects any token carrying `submissionScope`, as a backstop.
+- **Only explicit scoped requests get a submission token.** `api/AuthToken.java:88-89` mints one only when a valid scope is requested. A write token also requires the caller to be enrolled while admission is enabled.
+- **Tests cover both directions.** `SubmissionAuthenticationFilterTest.java:81-97` uses a real RSA key pair. It shows that `verify()` throws on a submission token and that the legacy search filter returns 401 without continuing the chain. `:50-54` shows an identity-only token gets 401 from the submissions filter.
+
+## The other round 1 items are resolved
+- **Limits:** the capabilities response and the schema now both include `maxDraftsPerUser` and `maxFieldsPerRow` (`Submissions.java:75-77`, `openapi.yaml:407-412`).
+- **Status codes:** PUT rows no longer lists 410, so non-editable drafts return 409 as documented. The If-Match pattern is now `{1,18}` in both the code and the spec.
+- **Retry-After:** `error()` no longer sends it, so the 20-draft quota 429 has none.
+- **Filter structure:** `chain.doFilter` now runs outside the authentication try/catch (`SubmissionAuthenticationFilter.java:79`).
+- **JSON parsing:** bodies are decoded as strict UTF-8, and Jackson is limited to 32 nesting levels. Duplicate keys, non-object roots and trailing content are rejected before org.json sees the text. The parser tests cover unquoted keys, single quotes, trailing values, duplicates, depth and invalid UTF-8.
+- **Entity:** the fields are private and the changes happen inside the entity's own methods (`submission/Submission.java:35-36`).
+- **Database tests** (`SubmissionStoreDbTest`):
+ - Rows survive a persistence restart.
+ - Competing edits produce exactly one 200 and one 412.
+ - An unconfirmed commit is reported as 503 and the rows are rolled back.
+ - An expired draft stays readable, rejects edits with 409, and an admin can read it.
+ - The quota race gives exactly one 201 and one 429. Because an expired draft for the same owner is already present, this also proves expired drafts don't count toward the quota.
+- **Error order:** admission is still checked before routing. That was a cosmetic item and you kept it deliberately, which is reasonable.
+
+## Remaining Minor issues
+1. **Strings containing U+0000 or unpaired surrogates pass validation but can't be stored cleanly.** `SubmissionJson` accepts escapes like `"\u0000"` and `"\ud800"` anywhere: source name, batchId, clientRowId, field keys and field values. PostgreSQL rejects NUL in text columns. That surfaces as a 500 from `makePersistent`, or as the 503 "Commit outcome unavailable; retry using the original operation key". A client that follows that advice retries the same bad input forever. An unpaired surrogate is instead silently replaced with `?` when written to the database. **Fix:** reject U+0000 and unpaired surrogates with 400 during parsing or in `requiredString` and the field checks, and add a test.
+2. **The published rows schema is looser than the code.** `SubmissionApiRows.rows` has no `maxItems: 200`, and `fields` has no `maxProperties: 256`. The code enforces both, returning 413 for too many rows and 400 for too many fields. Field-name length also has no limit beyond the 2 MB body cap. **Fix:** add the two bounds to the spec, and optionally a `propertyNames` max length.
+3. **The filter's 405 isn't documented.** `SubmissionAuthenticationFilter` returns 405 with `Allow` for methods like HEAD, OPTIONS and PATCH (using code `BAD_REQUEST`), but no operation in the spec lists 405. This is spec completeness only.
+
+None of these affect the stage 2 security boundary (ownership, scope, enrollment, keeping scoped tokens out of other endpoints) or the draft state rules. They can go into this change or be tracked separately. Uploads, validation, commit and workers remain gated and weren't reviewed.
diff --git a/docs/design/submissions/reviews/stage-3-disposition.md b/docs/design/submissions/reviews/stage-3-disposition.md
new file mode 100644
index 0000000000..d2a138b3e5
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-3-disposition.md
@@ -0,0 +1,17 @@
+# Stage 3 disposition
+
+Claude round 1 found four Majors, no Critical issues. Round 2 found no Critical
+or Major issues. Corrected multipart size/malformed errors and decode failures,
+aligned contract error codes, checked staging against webapps/import/upload roots,
+and bounded processing slots before database allocation. Upload/validation use
+nonblocking draft locks; image decoding is subsampled within bounded dimensions.
+
+Also rejected duplicate media and extension mismatches, guarded cleanup, and added
+unknown-length multipart, corrupt-image, overlap and processing-slot tests. Published
+contract corrections cover rowFields, 405/408, noneditable state responses and
+validation returning structured 200 reports. The one-slot pilot limit and container
+read-timeout requirement are documented in the runbook. One bounded retry candidate
+may temporarily exceed completed draft byte capacity; this is intentional.
+
+The combined stage 3–5 targeted run passed 37 tests, zero failures/errors/skips.
+Later queue/importer tests and the final integration build are recorded in README.
diff --git a/docs/design/submissions/reviews/stage-3-round-1.md b/docs/design/submissions/reviews/stage-3-round-1.md
new file mode 100644
index 0000000000..86e07c3ee7
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-3-round-1.md
@@ -0,0 +1,63 @@
+Stage 3 does not converge yet: I found **four Major issues**, all with local fixes. I read the listed sources, the tests, `UploadPaths`, `BulkImportUtil`/`BulkValidator`, `Shepherd` commit/close, `web.xml` and the target spec. I did not edit or run anything, and I'm making no claims about compile or test results.
+
+The rest holds up. Paths are contained: names are checked exactly, blob UUIDs are generated by the server, and symlinks are rejected. Streams are bounded and decoding is guarded by dimension limits. Retrying with the same content removes the duplicate blob and doesn't advance the revision. Owner and context checks go through `owned()` everywhere. Validation writes nothing to domain tables, doesn't change the revision, and `setFiles`/`replaceRows` clear the report. Every JDO field change happens inside `Submission` methods, so enhancement tracks it. The locking and the choice to keep the blob after an uncertain commit match what you decided.
+
+## Critical
+None.
+
+## Major
+
+**M1. Routine upload failures return 500 instead of 413/400/422.** `SubmissionFiles.java:49-65`, `Submissions.java:58`
+- **Oversize files.** In commons-fileupload 1.5, the `fileSizeMax` stream stops at exactly `limit` and throws `FileUploadBase.FileUploadIOException` (an `IOException` wrapping `FileSizeLimitExceededException`). That happens before `write()`'s own `count > limit` check can fire. So an oversize streamed file hits the generic `catch (Exception)` and becomes a 500. The same happens when a chunked request exceeds `sizeMax`. The servlet's `SizeLimitExceeded`/`FileSizeLimitExceeded` catch only sees these when the Content-Length is known up front.
+- **Malformed requests.** Malformed multipart, a missing boundary and `InvalidFileNameException` (NUL in the name) also become 500s.
+- **Bad images.** Truncated or undecodable images, and CMYK JPEGs, throw `IIOException`/`IOException` from `inspect()` and become 500s. The only case covered correctly is "no reader found", which returns 422.
+- **Why it matters:** the contract says to retry a 500 "only according to idempotency rules", so a mobile client may keep resending an oversize file.
+- **Fix:** in `receive()`, catch `FileUploadIOException` and unwrap its cause: size-limit causes become 413 `LIMIT_EXCEEDED`, and any other `FileUploadException` becomes 400 `BAD_REQUEST`. In `inspect()`, wrap the reader and decode calls so an `IOException` becomes 422. Add a servlet-level multipart test for an oversize chunked body and for a truncated PNG.
+
+**M2. Some error codes aren't in the contract's closed `Error.code` enum.** (`openapi.yaml:1025-1043`)
+- `SubmissionStore.java:121`: `FILE_NAME_CONFLICT` should be `FILE_CONTENT_CONFLICT`.
+- `SubmissionFiles.java:115`: `VALIDATION_FAILED` should be `BAD_REQUEST` (keeping status 422), or you could add a documented code.
+- `SubmissionStore.java:102`: the 410 returns `INVALID_STATE`, but the spec reserves `INVALID_STATE` for 409. Use `GONE`.
+- `SubmissionStore.java:123-124`: the contract says quotas are 429 and input size is 413. The draft file-count limit should therefore be 429 `LIMIT_EXCEEDED` with `Retry-After`, or you document it as 413.
+
+Clients generated from the enum will fail to deserialize these errors.
+
+**M3. The staging directory isn't checked against the document root or data directory.** `SubmissionFiles.java:29-40`
+
+Only `uploadTmpDir` is checked for overlap. You required staging to sit outside the document root too, but nothing stops it from being placed under the webapp's real path or `webapps/`, and Wildbook serves both statically.
+- **Fix:** pass `getServletContext().getRealPath("/")` from `Submissions` into `SubmissionFiles`. Then reject any root that overlaps (in either direction) the canonical webapp root, its parent `webapps` directory (which covers the data dir) or `CommonConfiguration.getImportDir`.
+- **Also:** create blob directories as owner-only (`PosixFilePermissions` 700) where the filesystem supports it.
+- **Tests:** add cases for each overlap.
+
+**M4. Nothing limits how many uploads can hold a database connection or how long they can hold it.** `SubmissionStore.java:105-115`
+
+Holding the draft lock for the whole stream is fine as you designed it, but each upload also keeps one pooled DataNucleus connection, which the whole app shares. The stream is bounded in bytes, not in time; Tomcat's read timeout resets on each packet. One enrolled user can hold 20 drafts, and slow mobile uploads to them in parallel can starve the webapp's connection pool.
+
+A second upload to the same draft is also a problem: it holds a connection while it waits up to 10 seconds on `pg_advisory_xact_lock`, then gets a 503.
+
+Decoding makes this worse: a 24 MP 16-bit RGBA PNG takes about 192 MB of heap, and neither upload nor validate limits how many decodes run at once.
+- **Fix, keeping your design:**
+ - Add a JVM-wide semaphore for uploads and validates, plus at most one in-flight upload per owner. Acquire them before `open()` using `tryAcquire`, and return 429 with `Retry-After` on failure.
+ - Enforce a wall-clock deadline in the `write()` loop.
+ - For the file lock, use `pg_try_advisory_xact_lock` and return 429.
+ - Consider decoding with `ImageReadParam.setSourceSubsampling`: the whole stream is still read, but memory is about 1/64.
+
+## Minor
+1. **Cleanup ordering.** In `SubmissionStore.java:127-131`, if `rollbackAndClose()` throws, `storage.remove` never runs. If `remove` throws an `IOException`, it hides the original 409/413. Nest the `try/finally` blocks and log removal failures instead of propagating them.
+2. **Orphaned blobs.** Blobs from uncertain commits, cancelled drafts and expired drafts are never collected. Log the blob id when a commit is uncertain, and plan a sweeper that deletes unreferenced blob directories older than the draft TTL.
+3. **Extension vs. content.** An `a.png` containing JPEG bytes is accepted as `image/jpeg`. Require `.png` for PNG and `.jpg`/`.jpeg` for JPEG, compared case-insensitively.
+4. **Media reference checks** in `SubmissionValidator.java:49-53`:
+ - A field like `Encounter.mediaAsset0.keywords` gets both `UNSUPPORTED_FIELD` and `MISSING_MEDIA`. Run the media check only when `supported(field)` matches the index regex.
+ - Duplicate references within a row are silently collapsed by the `Set`. Report them as an error.
+ - The same file referenced from several rows makes the importer attach one `MediaAsset` to several encounters (`BulkImporter.java:717`). Reject that or at least warn.
+5. **Capabilities.** `Submissions.java:84-89` advertises upload/validate even when staging isn't configured, which makes those calls return 503. Also, `maxDraftBytes` duplicates the constant; use `SubmissionFiles.MAX_DRAFT_BYTES`. `Receiver.receive(long remainingBytes)` is really passed the per-file maximum, so rename the parameter.
+6. **Validation under lock.** Validation re-hashes and decodes up to 200 files / 200 MB while holding the lock, which can push other writers past the 10-second lock timeout. The M4 semaphore covers most of this; keep it in mind for commit.
+7. **Test gaps.** Missing coverage for:
+ - multipart handling at the HTTP level
+ - the total draft-byte and file-count limits
+ - an upload whose commit is uncertain (blob kept, nothing reported as saved)
+ - `store.validate` leaving the revision unchanged and moving to `validated`, then back to `draft` after an upload or row replacement
+ - the staging-overlap checks
+ - rejection of over-large dimensions
+
+**Convergence:** no. Stage 3 converges once M1–M4 are fixed, with tests for the M1 multipart mapping and the M3 overlap checks. The Minor items can follow.
diff --git a/docs/design/submissions/reviews/stage-3-round-2.md b/docs/design/submissions/reviews/stage-3-round-2.md
new file mode 100644
index 0000000000..6a19da0b14
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-3-round-2.md
@@ -0,0 +1,64 @@
+Stage 3 now converges: M1–M4 are fixed and I found no Critical or Major issues. This was a read-only review. I didn't run anything, so compile and test results are whatever your current run reports.
+
+## Verification of M1–M4
+
+**M1: fixed.**
+- `receive()` at `SubmissionFiles.java:58-66` unwraps `FileUploadIOException`. A size-limit cause returns 413 and anything else returns 400. `FileUploadException` and `InvalidFileNameException` return 400, which covers errors raised by `hasNext()`.
+- `inspect()` at `:154` turns any `IOException` from reading or decoding into 422 `VALIDATION_INVALID`. It doesn't swallow the dimension 413, because `SubmissionException` is a `RuntimeException`.
+- `SubmissionFilesTest.java:39-49` exercises an oversize file and a truncated PNG with unknown length through the real parser, and checks that no staging files are left behind.
+
+**M2: fixed in code.** `FILE_CONTENT_CONFLICT` (`SubmissionStore.java:127`), `VALIDATION_INVALID` (`SubmissionFiles.java:108,139,154`) and 410 `GONE` (`SubmissionStore.java:102`) are all in the `Error.code` enum at `openapi.yaml:409-427`. The file-count limit returns 413.
+
+**M3: fixed.**
+- `configuredRoot` (`SubmissionFiles.java:29-45`) takes the real path of the staging directory and rejects overlap in either direction with:
+ - the legacy upload directory
+ - the parent of the webapp's real path, which covers the data directory
+ - `importDir`
+- It fails closed if `getRealPath` returns null or the parent is null. The only public constructor that reads configuration requires a `ServletContext`.
+- Blob directories are created with mode 700 where the filesystem supports POSIX permissions.
+- Tests cover both overlap directions.
+
+**M4: fixed.**
+- Upload and validate take the `SubmissionResources` slot (JVM-wide plus one per owner) before `open()`, and return 429 with `Retry-After: 5` when it's taken.
+- The draft lock uses `pg_try_advisory_xact_lock` and returns 429.
+- The write loop has a 2-minute wall-clock limit.
+- Decoding uses 4×4 subsampling.
+- The nested `finally` at `SubmissionStore.java:133-140` always runs blob removal and logs removal failures instead of throwing them.
+
+**Also verified:**
+- Duplicate media is rejected within a row and across rows (`SubmissionValidator.java:54-55`).
+- The media check only runs on supported fields, so the earlier double-report is gone.
+- The file extension must match the content (`SubmissionFiles.java:106-108`).
+
+## Critical
+None.
+
+## Major
+None.
+
+## Minor
+1. **Truncated multipart body still returns 500.** If the body ends mid-part, the part stream throws `MultipartStream.MalformedStreamException`. That's a plain `IOException`, not a `FileUploadIOException`. It escapes `receive()` and hits the generic 500 handler at `Submissions.java:60`, which logs a stack trace. Retrying is the right client behaviour anyway, and most cases are client disconnects, so this isn't Major. Fix: catch `MultipartStream.MalformedStreamException` in `receive()` and return 400. The "truncated" test covers a truncated PNG, not a truncated multipart body.
+2. **The JVM-wide slot count is 1** (`SubmissionResources.java:5`). One admitted user uploading slowly for 2 minutes, or validating a 200-file draft, which has no time limit, gets every other user a 429 for that whole time. That's acceptable for a gated pilot, but make it configurable and consider a separate, smaller limit around `inspect()` only, before wider rollout.
+3. **Unsupported formats return 422 with `CAPABILITY_UNAVAILABLE`** (`SubmissionFiles.java:143`). A GIF, BMP or TIFF has an ImageIO reader, so it reaches this line. The spec implies `CAPABILITY_UNAVAILABLE` means 503. Use `VALIDATION_INVALID`.
+4. **Limits are only checked after the whole stream is received.** When a draft already has 200 files, or no byte budget left, a new file is streamed in full before the 413 at `SubmissionStore.java:129`. Check the count and remaining bytes before `receive()` (a retry of an existing name is the exception) and pass `min(maxFileBytes, remaining)`.
+5. **`Retry-After` depends on the message text.** It's only set when the message contains "five seconds" (`SubmissionAuthenticationFilter.java:94`), so the 20-active-drafts 429 has no `Retry-After`. Put the header value on the exception instead.
+6. **Cleanup can hide the original error.** `receiveMultipart`'s cleanup (`:82`) and `write()`'s `finally` (`:113`) can throw `IOException` and hide the original exception. Log it instead.
+7. **Test gaps:**
+ - `DUPLICATE_MEDIA` (within a row and across rows) is not asserted in `SubmissionValidatorTest`.
+ - No extension-mismatch test.
+ - No test for the `tryLock` 429, the file-count limit or the draft-byte limit.
+ - No truncated-multipart-body test.
+ - `SubmissionFilesTest:72` holds the static global slot, so running JUnit tests in parallel would make the DB upload test fail intermittently.
+8. **Deployment note, as you said:** a connection that goes idle mid-read is only bounded by the container's read timeout. The deployment docs must require one.
+
+## Spec lag (minor schema issues, separate from the above)
+- **408 not documented:** the upload wall-clock limit returns 408 `BAD_REQUEST`, but 408 isn't listed for `uploadSubmissionFile`.
+- **405 not documented:** the filter's 405 isn't documented anywhere.
+- **410 never returned:** upload and validate list 410, but `editable()` returns 409 `INVALID_STATE` for cancelled or expired drafts. Either document 409, or run the 410 check before `editable()`.
+- **Validate 422 doesn't match:** the validate description says 422 means no rows. The code returns 200 with `valid=false` and `REQUIRED_VALUE`.
+- **`rowFields` shape:** the schema requires `additionalProperties: {type: object}`, but the code emits `supported` (array), `indexedMedia` (string) and `required` (array).
+- **Undocumented capabilities fields:** `maxFiles`, `maxImagePixels`, `uploadMediaTypes` and `operations` values aren't in the schema. They're allowed by `additionalProperties: true` but not described.
+- **200-file limit wording:** I couldn't find the 413 for this limit stated explicitly. Only the generic "413 for input size" line at `openapi.yaml:440` covers it. Add it to the upload description or add a `maxFiles` limit.
+- **Undocumented issue codes:** `DUPLICATE_MEDIA`, `MISSING_MEDIA` and `INVALID_MEDIA` aren't listed. `Issue.code` is free-form, so nothing breaks.
+
+As you asked, I didn't review the Stage 4/5 worker and commit code (`SubmissionJobs`, `SubmissionImporter`) or their tests.
diff --git a/docs/design/submissions/reviews/stage-4-disposition.md b/docs/design/submissions/reviews/stage-4-disposition.md
new file mode 100644
index 0000000000..516a316380
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-4-disposition.md
@@ -0,0 +1,18 @@
+# Stage 4 disposition
+
+Claude round 1 found a Critical implicit-transaction-commit problem in the legacy
+Shepherd save helpers. Deferred mode now persists encounters, occurrences,
+individuals and projects directly through the caller's PersistenceManager. Legacy
+mode retains its helpers. ImportTask saves also use direct persistence so errors
+propagate. Deferred mode skips independent progress transactions, cache eviction
+and derivative/index dispatch; raw row diagnostics are suppressed on this path.
+
+Round 2 found no Critical or Major issues and traced all reachable calls for the
+strict pilot field subset. Added explicit helper/dispatch boundary assertions and
+a real PostgreSQL rollback test for the deferred importer. The adapter revalidates
+approved revision/digests, resolves owner, rejects missing usernames, reuses media
+creation and records source-row mappings from actual row resolution.
+
+Legacy grouped-row/year-precision/missing-media fixtures remain in the targeted
+regression run. The combined run passed 37 tests; the later real importer rollback
+check and full integration build are recorded in README.
diff --git a/docs/design/submissions/reviews/stage-4-round-1.md b/docs/design/submissions/reviews/stage-4-round-1.md
new file mode 100644
index 0000000000..cd78305463
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-4-round-1.md
@@ -0,0 +1,56 @@
+I found **one Critical issue**. The rest of the adapter boundary looks sound. I only read the code; I didn't run any tests.
+
+## Critical
+
+**1. The deferred path still commits the caller's transaction.** The change skips `markProgress` and `updateStandardChildrenBackground`, but the persistence loop in `createImport()` still goes through Shepherd helpers that commit on their own:
+
+- `BulkImporter.java:155` calls `myShepherd.storeNewEncounter(enc, enc.getId())`. At `Shepherd.java:141-156` that runs `beginDBTransaction()`, which joins the caller's active transaction, then `pm.makePersistent`, then `commitDBTransaction()`.
+- `storeNewOccurrence` (`Shepherd.java:197`), `storeNewMarkedIndividual` (`:273`) and `storeNewProject` (`:374`) do the same thing.
+
+What this does in `SubmissionImporter.execute`:
+- The first encounter commits every MediaAsset and User saved so far, plus that encounter.
+- Each later encounter or occurrence opens and commits its own transaction.
+- The final `storeNewImportTask`, which marks the task complete, runs in a transaction that the helpers reopened. The caller's rollback only covers that last piece.
+
+Two more problems follow:
+- The helpers catch persistence errors, roll back, and return `"fail"` or `false`. So a failed encounter insert doesn't abort the import. The adapter then writes a mapping for rows that may not exist, and the task still ends up `complete`.
+- `BulkImporterSubmissionBoundaryTest` can't see any of this. `Shepherd` is a mock, so `storeNewEncounter` does nothing, and `verify(sh, never()).commitDBTransaction()` passes regardless.
+
+**Fix** (in `BulkImporter`, deferred mode only, so legacy callers keep the old behavior):
+```java
+// encounters
+if (deferSideEffects) { enc.setEncounterNumber(enc.getId()); myShepherd.getPM().makePersistent(enc); }
+else myShepherd.storeNewEncounter(enc, enc.getId());
+// occurrences
+if (deferSideEffects) myShepherd.getPM().makePersistent(occ); else myShepherd.storeNewOccurrence(occ);
+// individuals / projects: same pattern
+```
+With this, persistence errors propagate as a `ServletException` and the caller rolls back.
+
+In the test, add `verify(sh, never()).storeNewEncounter(any(), any())` and `verify(sh, never()).storeNewOccurrence(any())`, plus `verify(pm, atLeastOnce()).makePersistent(any(Encounter.class))`. Also add one legacy assertion that the `storeNew*` helpers are still called.
+
+## Major
+
+None, once Critical #1 is fixed.
+
+## Minor
+
+1. **Duplicate media references aren't rejected.** The validator builds `media` as a `Set` (`SubmissionValidator.java:46-57`), so:
+ - `mediaAsset0` and `mediaAsset1` can point to the same file. The encounter then gets two exemplar annotations on one MediaAsset, and the limit check counts that file once.
+ - Two rows can reference the same file. The importer makes one MediaAsset, and both encounters annotate it, so the mapping reports the same `mediaAssetIds` for both rows. That mapping is accurate, but it's probably not what "separate encounter each row" is meant to imply.
+
+ Fix: in the validator, reject a repeated value within a row, and either reject or knowingly allow reuse across rows (e.g. a `DUPLICATE_MEDIA_REFERENCE` issue).
+2. **An owner without a username fails late.** If `owner.getUsername()` is null, the failure only shows up inside `processRow` ("no value for Encounter.submitterID") as a generic error. Add a check next to the eligibility check at `SubmissionImporter.java:17-18`: `Util.stringIsEmptyOrNull(owner.getUsername())` → `SubmissionException(403, "ACCESS_DENIED", …)`.
+3. **The stale-validation check doesn't pin identity.** It compares the digests and normalized rows but not `approved.submissionId` or `approved.revision` against the draft. The content comparison covers most of the risk, but pinning these two is cheap: add `approved.getString("submissionId").equals(draft.getId()) && approved.getInt("revision") == draft.getRevision()`.
+4. **Some test gaps:**
+ - The boundary test passes `importTaskId = null`, so it never tests that `markProgress` is suppressed. Use a non-null ID and `mockConstruction(Shepherd.class)`, then assert nothing was constructed.
+ - There's no adapter-level test showing that `clientRowId` → encounter/occurrence/media mapping follows row order.
+
+## Checked and fine
+
+- **Legacy behavior:** the default is `deferSideEffects = false` with a null collector. `processRow` only gained a return value, and progress and dispatch are unchanged for existing callers.
+- **No background work on the new path:** child-image generation and OpenSearch indexing are both inside the `!deferSideEffects` guard, and `markProgress` returns early.
+- **One encounter per row:** the strict field set has no `Encounter.id`/`catalogNumber`, sighting ID or individual ID. So each row gets a new UUID encounter and its own `Occurrence`, and no existing individual is touched.
+- **ID mapping:** row IDs come from `processRow` through the collector, keyed by row index, and are read after persistence. MediaAsset uses `value-strategy="identity"`, and legacy code already reads `getIdInt()` right after `save`, so the IDs are there.
+- **Missing media:** `UploadedFiles.makeMediaAsset` throws instead of returning null. Together with the `media.keySet().equals(requiredFiles)` check, that means BulkImporter's silent skip of a missing asset (`BulkImporter.java:717-731`) can't be hit from this path.
+- **Authorization and copies:** the owner comes from the draft (the server-side owner ID), and eligibility is rechecked at execution. `submitterID` is injected, so the client can't set it. Staging paths are UUID-blob plus checked-name with containment and symlink checks. Filenames are already `cleanFileName`-idempotent and unique ignoring case (`SubmissionStore.java:118-121`), so copies into `Encounter.subdir(taskId)` can't collide. Leftover files on rollback are the orphan case you've already deferred to the worker/operator stage.
diff --git a/docs/design/submissions/reviews/stage-4-round-2.md b/docs/design/submissions/reviews/stage-4-round-2.md
new file mode 100644
index 0000000000..8192b3ca7b
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-4-round-2.md
@@ -0,0 +1,50 @@
+**This round converges: no Critical and no Major findings.** I only read the code and didn't run anything. The claims below come from reading the code, not from test results.
+
+## Critical #1 from round 1: resolved
+
+- **Persistence loop:** in deferred mode, the loop at `BulkImporter.java:153-196` now calls `pm.makePersistent` directly for encounters, occurrences, individuals and projects. `setEncounterNumber` is kept, matching `Shepherd.storeNewEncounter:142`. MediaAssets (`MediaAssetFactory.save:81`) and Users (`:148`) were already direct `makePersistent` calls, so nothing on this path begins, commits or rolls back a transaction.
+- **Legacy path:** the `else` branches still call the `storeNew*` helpers, `cacheEvictAll` and `updateStandardChildrenBackground`, so legacy behavior is unchanged.
+- **Progress updates:** `markProgress` (`:760`) returns early in deferred mode, so its separate Shepherd, its commit and its `catch`-and-print are never reached.
+- **Error propagation:** exceptions from `processRow` are wrapped in a `ServletException` (`:114-119`). Exceptions from `makePersistent` in the persistence loop are not caught in `createImport`. Either way they reach `SubmissionJobs.execute:81`, which rolls back and calls `fail(..., "IMPORT_FAILED")`.
+- **ImportTask writes:** both call sites now use `sh.getPM().makePersistent(task)` (`SubmissionImporter.java:60`, `SubmissionJobs.java:52`). Neither swallows errors, and the task only reaches `complete` inside the caller's transaction.
+
+## Other commit or swallow paths in the strict field subset
+
+I followed every call reachable from the 16 fields in `SubmissionValidator.FIELDS` plus `mediaAssetN` and the injected `submitterID`:
+
+| Path | Result |
+|---|---|
+| `getOrCreateMarkedIndividual(null, …)` | Returns null right away. Individual, social-unit and name code is unreachable. |
+| `Shepherd.getOrCreateOccurrence(null)` (`Shepherd.java:3096`) | Creates a new object in memory only. |
+| `getOrCreateEncounter` | No ID fields, so it creates a new UUID encounter with no lookup. |
+| Submitter, photographer, inform-other, project, measurement and sample loops | No matching fields, so none of them run. The swallowing `catch` in `handleSocialUnit` can't be reached. |
+| `handleKeywords` | Called with an empty set, so it does nothing. |
+| `new Annotation(tx, ma)`, `addEncounterAndUpdateIt`, `setLatLonFromEncs`, `setSubmitterIDFromEncs` | In memory only. |
+| `UploadedFiles.makeMediaAsset` / `AssetStore.getDefault` | Reads and file copies only. Failures throw `ApiException`; nothing returns null. |
+| `BulkValidator` / `validateRow` | Reads only (`getUser`, taxonomy and config lookups). |
+| `SubmissionPolicy.enrolled` | Reads config only. |
+| `bulkOpensearchIndex` (the only other `new Shepherd` in the bulk package) | Only reached from the legacy dispatch. |
+
+I found no remaining hidden commits, and no persistence errors that get swallowed.
+
+## Major
+
+None.
+
+## Minor
+
+1. **An owner without a username still fails late** (round-1 Minor #2, not addressed). If `owner.getUsername()` is null, `fields.put("Encounter.submitterID", null)` at `SubmissionImporter.java:29` removes the key. The import then fails inside `processRow` with a generic `IMPORT_FAILED` instead of `ACCESS_DENIED`. It fails safely, but the reason is unclear. The fix is one guard next to lines 17-18.
+2. **The boundary test has a few gaps** (`BulkImporterSubmissionBoundaryTest.java`):
+ - It passes a non-null task ID, but `markProgress` suppression is only covered indirectly: if suppression broke, a real `new Shepherd` would presumably throw. A `mockConstruction(Shepherd.class)` asserting zero constructions would make this explicit.
+ - It doesn't verify `never().cacheEvictAll()` before the legacy run, or `atLeastOnce().cacheEvictAll()` after it.
+ - It doesn't verify `pm.makePersistent(any(Occurrence.class))` alongside the Encounter check.
+3. **There's still no adapter-level test for row-order mapping** (round-1 Minor #4b). Nothing checks that `clientRowId` maps to the right encounter, occurrence and media for each row.
+
+## Checked and fine
+
+- **Stale-validation check (`:21`):** it now compares `valid`, `revision`, both digests and the canonical rows. `submissionId` isn't compared, but the validation JSON is written by the server onto the same draft, so that's acceptable.
+- **Duplicate media:** repeats within a row and reuse across rows are both rejected as `DUPLICATE_MEDIA` (`SubmissionValidator.java:54-55`). That closes round-1 Minor #1.
+- **Media availability:** the `media.keySet().equals(requiredFiles)` check plus `makeMediaAsset` throwing means the silent skip at `BulkImporter.java:722` still can't be hit.
+- **Orphaned staged files on rollback:** still deferred to Stage 5, as agreed.
+
+**Convergence:** the transaction boundary for the strict field subset is sound. The remaining items are Minor and don't block. The Stage 5 worker is outside this review.
diff --git a/docs/design/submissions/reviews/stage-5-disposition.md b/docs/design/submissions/reviews/stage-5-disposition.md
new file mode 100644
index 0000000000..289fe886c1
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-5-disposition.md
@@ -0,0 +1,9 @@
+# Stage 5 review disposition
+
+Four read-only Claude rounds reviewed commit, worker, persistence, recovery and cleanup. See the numbered transcripts. Round 4 found no remaining Critical or Major issues.
+
+Corrections include separating pending work from history, keyset replay, derivative-specific recovery timestamps, short independent cleanup transactions, durable manifest-release progress, a bounded inventory deadline, per-owner daily admission, and preserving uncertain execution for operator reconciliation. Cleanup errors do not block intake.
+
+Remaining minor observations: released terminal manifests are empty (documented); a failed database cleanup operation safely defers physical deletion to a later pass; pagination/deadline and certain-failure cleanup deserve additional scale coverage. Operator completion timestamps and context-specific staging are now documented. The pilot table requires schema verification before first deployment.
+
+Claude performed source review only. Executed test evidence and deployment gates are recorded in ../README.md and ../pilot-runbook.md.
diff --git a/docs/design/submissions/reviews/stage-5-round-1.md b/docs/design/submissions/reviews/stage-5-round-1.md
new file mode 100644
index 0000000000..cf97b78ff9
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-5-round-1.md
@@ -0,0 +1,87 @@
+# Stage 5 review: submission queue, worker, importer and lifecycle
+
+This was a read-only review. I didn't run any builds or tests, and I edited nothing.
+
+**Verdict:** no Critical findings. There are 2 Major findings and 11 Minor ones. Most of the properties you listed hold:
+
+- **Acceptance before 202:** the ImportTask and the queued draft commit in one transaction before the 202, and a lost response replays.
+- **One execution regardless of keys:** the per-submission advisory lock plus the `jobId` check allow only one accepted execution.
+- **Claims across instances:** a worker-wide lock plus the "any `importing` row" guard enforce one active import across the installation.
+- **Revalidation before import:** owner, enrollment, config digest and manifest are rechecked before anything is copied.
+- **Caller-owned transaction:** in deferred mode the bulk importer only calls `makePersistent`, with no hidden commits or progress writes.
+- **Atomic import result:** domain objects, `imported(result)` and ImportTask completion commit together.
+- **Uncertain-commit reread is fenced:** `fail()` blocks on the old session's advisory lock, so a commit that actually landed shows as `imported`.
+- **Derivatives only after domain commit.**
+- **Lifecycle:** the worker is off by default and stops on shutdown.
+- **Admission off still drains accepted work:** the worker only checks `workerEnabled`.
+- **Unenrolled owners** fail with the terminal `failed` state before any copy.
+- **Strict field subset:** unsupported fields are rejected, and `Encounter.id`, sighting and individual keys can't get in.
+- **No automatic AI dispatch.**
+- **Owned reads** skip the admission check in both the filter and the servlet.
+
+## Major
+
+**M1. The postprocess scan can permanently skip new imports, and restart replays all history** (`SubmissionJobs.java:129-131`, `SubmissionWorker.java:34-37`)
+- **Problem:** the scan query matches `derivatives == 'complete'` with no filter on `phase`. It takes the oldest 10,000 rows first, and done items are only skipped via the in-memory `dispatched` set.
+ - Once about 10,000 finished submissions exist, rows that still need derivatives fall outside the window and never get derivatives or indexing. Nothing reports it.
+ - After every restart, and in every JVM, each tick replays indexing for every finished submission. The whole loop runs while holding the JVM's single resource slot, so uploads and validation get 429 until it finishes.
+ - `dispatched` grows without limit.
+- **Fix:**
+ - Split it into two queries. Do the real work first: `derivatives == 'pending' || (derivatives == 'complete' && phase == 'pending')`.
+ - Do replay separately and bounded: `phase == 'unknown'` limited by a recent import-time window or a per-JVM start marker, processed in batches (for example 50 per tick).
+ - Replace the unbounded `dispatched` set with a replay marker, or cap it.
+
+**M2. A second JVM can wrongly mark derivatives `unknown` because staleness uses the import-claim time** (`SubmissionJobs.java:142-149`, `190`, `199`)
+- **Problem:** `postprocess` commits `derivatives = running`, releases the lock, then takes the lock again in a new transaction. `reconcileStaleClaims` judges `running` rows by `workStartedAt`, which is the time of the import claim.
+ - If postprocessing starts more than an hour after the claim (worker toggled off, restart, or backlog), another JVM's reconcile can grab the lock in that gap and mark the derivatives `unknown`.
+ - The derivatives are then held for manual reconciliation even though nothing was interrupted.
+- **Fix:** add a `derivativesStartedAt` field, set it in `derivatives("running")`, and use it in the reconcile filter. Alternatively, use a claim token that the second transaction checks.
+
+## Minor
+
+1. **One bad item stops the rest of the tick** (`SubmissionWorker.java:34-38`). An exception in one `postprocess` aborts the whole loop, and the same oldest item fails again every tick. Wrap each item in its own try/catch and log it with the item's id.
+2. **The ImportTask is checked too late** (`SubmissionImporter.java:57-58`). A missing ImportTask (for example, deleted from the legacy task UI) is only detected after the asset copies. That leaves copied files behind and marks the job uncertain when it could have simply failed. Move the check before `makeMediaAsset`.
+3. **Error classification relies on exception type** (`SubmissionJobs.java:84`). Anything thrown as a `SubmissionException` counts as "no side effects". That's true today, but only by convention.
+ - A 503 lock timeout in `execute` before `run` becomes a terminal `failed`, and the draft can't be committed again.
+ - A validation JSON missing a digest key throws a `JSONException`, which becomes `needs_reconciliation`.
+ - Fix: add a dedicated pre-import rejection type thrown only before the first copy. If the lock or open fails before the attempt starts, leave the row `importing` rather than failing it.
+4. **Reconcile leaves the ImportTask at `queued`** (`SubmissionJobs.java:198`). `fail()` updates the ImportTask status but reconcile doesn't, so legacy task views show a live task. Set the task to `needs_reconciliation` there too.
+5. **A crashed import blocks the whole queue for an hour.** One JVM crash mid-import stops all claims installation-wide until reconcile runs 60 minutes later. Document this, or cut the threshold for claims whose lock `tryLock` finds free (the owning session has died).
+6. **Uncertain claim commit** (`claimNext`, `SubmissionJobs.java:67-68`). If the claim's commit result is unknown, the row may be `importing` with no JVM running it. It's held rather than rerun, which is safe but blocks the queue until reconcile. A `claimToken` would let the claimer re-read and continue.
+7. **Cleanup inventory** (`SubmissionJobs.java:208-211`).
+ - It scans every submission in the context and stops for good once there are more than 10,000, with no log message.
+ - Blobs for `failed` and `imported` drafts are never deleted, so private staging duplicates the asset store forever.
+ - Fix: only fetch rows whose blobs must be kept (not cancelled, not expired) and page through them. Log when cleanup refuses to run. Document the retention policy for `imported` and `failed`.
+8. **Expiry race** (`SubmissionJobs.java:214`). A draft committed just before it expires can have its files deleted by a cleanup that read the inventory just before the commit. The importer's revalidation turns this into `failed`, so no data is lost. Excluding drafts that expired less than a day ago would close the window.
+9. **Worker startup and resource slot.**
+ - The worker is built during startup only when `workerEnabled` is already true, so turning it on at runtime needs a restart.
+ - It starts before OpenSearch and the IndexingManager are initialised; the first tick is 10 s later, so this is only a delay risk.
+ - The worker shares the single per-JVM slot with uploads, so steady upload traffic can starve it; its 429s are silent.
+ - Fix: start the worker at the end of `contextInitialized`, and log repeated 429 skips.
+10. **Staging overlap checks and permissions** (`SubmissionFiles.java:29-45`).
+ - The overlap checks don't cover LocalAssetStore roots, which may be served outside webapps.
+ - Directory permissions are set after creation, and files keep the umask default.
+ - Fix: reject overlap with asset store roots, and create directories and files with owner-only permissions atomically.
+11. **Missing indexes** (`package.jdo`). There are none on `(context,state)`, `(context,ownerId,state)` or `jobId`, so every tick's queries scan the whole table.
+
+## Tests
+Gaps in `SubmissionStoreDbTest`, based on reading the source:
+- Two `claimNext` calls running concurrently: the current "one claim" test calls it twice in sequence.
+- A commit that lands but reports failure (a subclass that commits, then returns `false`), asserting the row ends up `imported`, not `needs_reconciliation`.
+- `reconcileStaleClaims`: an `importing` row older than an hour becomes `needs_reconciliation`, and a locked row is skipped.
+- `postprocess`: a failure mid-generation leaves `running` and nothing reruns; a replay after restart re-queues indexing only.
+- `cleanup`: it refuses above the cap, and it keeps blobs for active, uncertain and imported drafts.
+- Execution with an unenrolled owner ends `failed` with no asset copy.
+- `readyJob`'s validation JSON has no `configDigest` or `manifestDigest`, so no DB test drives the real `SubmissionImporter`.
+
+## Contract mismatches (for stage 6)
+- `results.links.importTask` points to `/import/importTask.jsp`, which doesn't exist anywhere under `src/main` (probably removed with WebImport). The React route is `/react/bulk-import-task?id=`.
+ - That page may offer detection or identification actions. Check it before linking, to keep "no AI dispatch" true.
+- Capabilities advertise `idempotencyRetentionSeconds` = 7 days, but create and commit keys are replayed forever.
+- A lost-response commit replay after admission is turned off returns 503 `ADMISSION_DISABLED`, not the original 202. Clients should fall back to GET status.
+- Enabling the worker at runtime needs a restart; stopping takes effect on the next tick.
+- State values to document:
+ - `derivatives.state`: `pending | running | complete | unknown`
+ - `indexing.state`: `pending | unknown`
+ - error codes: `IMPORT_FAILED`, `COMMIT_OUTCOME_UNCERTAIN`, `INTERRUPTED_EXECUTION`
+ - GET submission omits `derivatives`; only results includes it.
diff --git a/docs/design/submissions/reviews/stage-5-round-2.md b/docs/design/submissions/reviews/stage-5-round-2.md
new file mode 100644
index 0000000000..7cff3a499b
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-5-round-2.md
@@ -0,0 +1,55 @@
+I found 1 new Major issue and no Critical ones. Both earlier Majors (M1 and M2) are fixed. This was read-only: I ran nothing, edited nothing, and haven't seen any build or test results.
+
+## Your corrections, checked against the code
+
+| Item | Status |
+|---|---|
+| M1: split pending and replay queries | ✅ `SubmissionJobs.java:130`: the pending query only matches rows with derivatives `pending`, or `complete` with indexing `pending`. The limit is 10. `replayBatch` (`:137-147`) only matches `complete` with indexing `unknown`, with `createdAt <= startup`, a stable `createdAt, id` order and pages of 5. The unbounded `dispatched` set is gone. |
+| Worker: pending first, per-item catch, one replay pass | ✅ `SubmissionWorker.java:37-50` |
+| M2: separate derivative timestamp | ✅ `Submission.java:55` sets `derivativesStartedAt` when derivatives go `running`, and the reconcile filter uses it (`SubmissionJobs.java:202`). |
+| ImportTask checked before copying files | ✅ `SubmissionImporter.java:29-30`, thrown as `PreImportRejection` |
+| Reconcile updates the ImportTask | ✅ `SubmissionJobs.java:212` |
+| Worker starts at the end of startup | ✅ `StartupWildbook.java:227` |
+| Busy (429) skips are logged | ✅ at most once a minute (`SubmissionWorker.java:52`) |
+| Cleanup: try-lock, refresh, keep busy drafts | ✅ This closes the old expiry race (Minor 8). It also causes the new Major below. |
+| Refusal above the cap is logged | ✅ |
+| Results link | ✅ It now points to `/react/bulk-import-task?id=`. I didn't recheck whether that page offers detection or identification actions. |
+| Indexes | ✅ `package.jdo:6-8` |
+
+## Major
+
+**N1. A cleanup failure stops all imports, and the new cleanup locks make that failure likely** (`SubmissionJobs.java:220-241`, `SubmissionWorker.java:29-32`)
+- **Lock build-up:** each expired or cancelled draft gets its own advisory lock, all in one transaction, and none is released until the whole inventory has been read.
+ - Expired and cancelled rows are never deleted, so their number only grows.
+ - PostgreSQL keeps advisory locks in a shared table sized by `max_locks_per_transaction × (max_connections + max_prepared_transactions)`. With default settings that's about 6,400 locks for the whole server. Below the 10,000-row cap, cleanup can hit `out of shared memory`, and while it holds all those locks, other sessions' lock requests can fail too.
+- **What happens after a failure:**
+ - The SQL error becomes a 503 `SubmissionException`. Any `IOException` from `storage.cleanup` (for example, permission denied on one blob) has the same effect.
+ - Either one ends the tick before `claimNext`, and `lastCleanup` is never updated. So cleanup runs and fails again on every 10-second tick.
+ - Result: work that already got a 202 never runs, and postprocessing and replay stop too. The only symptom is a repeating "requires inspection" / "unavailable" log line.
+- **Fix:**
+ - Set `lastCleanup` before calling cleanup, and wrap cleanup in its own try/catch so it can never block claiming.
+ - Delete blobs one directory at a time and catch errors per directory.
+ - For each expired or cancelled candidate, either use a short transaction per draft, or take a session lock with `pg_try_advisory_lock` and release it with `pg_advisory_unlock`, so locks don't pile up.
+
+## Minor
+
+1. **Replay uses the draft's creation time and re-queues all history on every start.**
+ - `createdAt` is when the draft was created, not when it was imported or dispatched. So submissions imported after startup can still be replayed (harmless duplicates).
+ - More importantly, each JVM start re-queues indexing for every imported submission ever, 5 every 10 s, and every JVM does it.
+ - The indexing queue is in memory (`IndexingManager.java:29`), so only rows dispatched shortly before the previous shutdown can have lost entries.
+ - Fix: add a `dispatchedAt` field, set it with `phase("unknown")`, and only replay rows where `dispatchedAt` falls in a window before startup.
+2. **Bad rows can fill the pending window.** A row whose derivatives are `complete` but whose indexing step keeps throwing stays `complete`/`pending` and is picked up every tick. Ten such rows fill `setRange(0, 10)` and hide newer work. A persistently failing derivative step doesn't have this problem, because it stays `running` and reconcile later marks it `unknown`. Fix: add an attempt count or a last-attempt time and skip rows that keep failing.
+3. **The worker still holds the JVM's single intake slot during derivative work** (from the old Minor 9). One tick can generate derivatives for 10 submissions × up to 200 images while holding the slot and a stage-2 transaction. Uploads and validation get 429 for the whole time. Consider not taking the slot for postprocessing and replay, or processing one submission per tick.
+4. **Cleanup inventory.** It still loads full `Submission` rows, including `rowsJson`, which can be up to 2 MB each, for up to 10,001 rows in one persistence manager. The cap counts terminal rows, which are never deleted, so it will eventually refuse permanently. Blobs for `imported` and `failed` drafts are still kept forever. Fix: fetch only the fields cleanup needs, use pages, and decide whether rows expire and how long blobs are kept.
+5. **Lock timeout before an import starts** (`SubmissionJobs.java:83`). The row stays `importing`, which blocks the installation-wide queue for an hour. Reconcile then marks it `needs_reconciliation` even though nothing ran. That's safe and matches the old Minors 5 and 6, but it should be documented as a known limitation.
+6. **Schema:** `derivativesStartedAt` is a primitive `long` added to an existing table. On any dev or staging database that already has `SUBMISSION` rows, check how DataNucleus schema auto-update adds the column (NOT NULL without a default).
+
+## Test gaps in the new DB tests
+- `staleClaimsAreHeld…` (`SubmissionStoreDbTest.java:251`) only checks the negative case: a fresh derivative claim isn't held. Nothing checks that an old `derivativesStartedAt` becomes `unknown`, or that a locked row is skipped.
+- `pendingWork…IsBounded` (`:263`):
+ - It never tests the startup watermark (it passes `now`) or that paging advances with the offset.
+ - The `work.contains(pending)` check depends on fewer than 10 older pending rows existing in the shared container. Other tests, such as `lostCommitAcknowledgment…` and the real-adapter test, leave imported/pending rows behind. That makes it order-dependent. Filter by the test's own IDs or use a fresh context.
+- `lostCommitAcknowledgment…` runs the steps one after another, so it proves that "imported wins" but not the lock fence during an in-flight commit. That's acceptable, but name it accordingly.
+- Still no cleanup tests (refusal at the cap, keeping blobs, busy drafts retained, lock build-up with many expired rows).
+
+The strict field subset is unchanged, as you said. The asset-root and atomic-permission Minor is left for you as agreed.
diff --git a/docs/design/submissions/reviews/stage-5-round-3.md b/docs/design/submissions/reviews/stage-5-round-3.md
new file mode 100644
index 0000000000..44f3cab5b3
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-5-round-3.md
@@ -0,0 +1,36 @@
+I found no Critical issues. N1's specific problems are fixed, but one Major issue remains: nothing ever deletes submission rows, so the new cleanup bound will eventually stop cleanup for good.
+
+## N1 status: fixed as described
+
+What I checked, in `SubmissionJobs.java:232-265` and `SubmissionWorker.java:29-33`:
+
+- **No long-held transaction or lock pile-up.** Each page of 100 rows is read in its own transaction, which is closed before any row is processed. Each candidate row (cancelled, or an expired draft) gets its own short transaction that tries the row's lock without waiting, re-reads the row, and releases. The worker's lock is never held across the whole pass.
+- **Rows that are busy or not yet final keep their files.** If the lock is taken, the file list read earlier is kept. The re-read also handles a race with commit: an `enqueue` that commits after expiry turns the row into `queued`, and its files are kept.
+- **Paging is safe.** Nothing deletes `Submission` rows, and new rows only push existing rows later in the order. The worst case is reading a row twice, which is safe; no row can be skipped.
+- **Stopping early deletes nothing.** Hitting the 10,000-row bound or the 10-second deadline returns before `storage.cleanup`. The file-cleanup step only removes directories that are old, not symlinks, contain only regular files, and whose files are also old. An error on one directory doesn't stop the others.
+- **The worker changes are in place.** It records the cleanup time before running cleanup, catches cleanup failures separately, and still claims work afterwards. Failed indexing moves to `phase=failed` and leaves the pending list. Reruns are limited to rows with `phase=unknown`.
+- **The DB test covers the main behaviours.** Blobs of draft, imported and `needs_reconciliation` rows are kept. A cancelled row's blob is kept while its lock is held and removed after release. An old orphan is removed on the first pass.
+
+## Major: once row count passes the bound, cleanup stops for good
+
+The count only ever grows:
+
+1. **The row count only grows.** Cancelled, expired, imported and failed rows are never deleted. The paging query (`SubmissionJobs.java:237`) counts every row in the context, so normal use eventually passes 10,000 rows. After that, cleanup logs "paused" every hour and never deletes anything again.
+2. **Every pass re-checks every old cancelled or expired row.** Nothing records that a row's files were already released, so each of these rows costs a lock transaction on every pass (`:247-248`), and every pass starts again at offset 0. As they pile up, the 10-second deadline (`:245`) starts firing before the end is reached, well before 10,000 rows.
+3. **One user can cause this on purpose.** Only *active* drafts count toward the 20-per-account limit (`SubmissionStore.java:36-41`), and cancelling frees the slot at once. Repeating create → upload up to 200 MB → cancel builds up rows until cleanup stops. After that, every blob uploaded and cancelled stays on disk forever. "Fail closed" then means the staging disk slowly fills up.
+
+A related gap: blobs of `imported` and `failed` rows are kept forever by design (`:249-250`, and the test asserts it). So even below the bound, staging only grows with each import, and there is no way to release them.
+
+**Suggested fix, a small change:**
+- Once a cancelled or expired row has been checked under its lock, write `filesJson = "[]"` in that same short transaction and commit it.
+- Filter the paging query to rows that still hold files (`filesJson != '[]'`).
+
+The bound and deadline then apply only to rows that actually hold blobs, and each old row costs one transaction in total. Separately, decide on a release rule for `imported` and `failed` blobs (for example, some time after `imported` with derivatives complete), or document that operators must clear staging by hand. Rate-limiting create+cancel per owner would close the deliberate route.
+
+## Minor
+- **`SubmissionFiles.cleanup` (`:179-197`) has no time limit.** The 50-directory cap counts only deletions, so the whole directory listing is still scanned and checked. That listing grows without limit because of the kept imported blobs.
+- **`reconcileStaleClaims` (`SubmissionJobs.java:209-219`) still collects up to 20 row locks in one transaction.** It's limited and brief, but it's the same pattern N1 removed from cleanup. If it throws, it also aborts the whole tick, including cleanup, because it runs first.
+- **Certain failures keep their files.** A `failed` row that came from a `PreImportRejection` could release its blobs instead of keeping them like the uncertain states.
+- **The test runs against the shared `context0` database with rows from other tests.** It's fine now, but the assertion that the orphan is removed would silently depend on the total row count staying under the bound.
+
+This was a read-only review: I ran no commands and didn't run the tests or build, so I haven't seen the full build's results.
diff --git a/docs/design/submissions/reviews/stage-5-round-4.md b/docs/design/submissions/reviews/stage-5-round-4.md
new file mode 100644
index 0000000000..dad1bca360
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-5-round-4.md
@@ -0,0 +1,26 @@
+I found no remaining Critical or Major issues in SubmissionJobs, SubmissionStore, SubmissionWorker, SubmissionFiles, SubmissionImporter, the `Submission` entity, `package.jdo` or `SubmissionStoreDbTest`. The round-3 cleanup fix works as described. This was read-only: I ran no commands or builds, so none of the tests have been run.
+
+## Confirmed fixed
+
+- **Cleanup no longer gets permanently stuck.** `SubmissionJobs.java:237-261` pages through rows with non-empty `filesJson` in `id` order, 100 at a time, with no row cap. Each release takes its own short lock and commits (`:263-276`). A locked row is skipped and its files kept. Released rows drop out of later passes.
+- **Deleting files on a partial inventory is safe.** When the deadline hits, the pass stops before deleting anything. Released references stay saved, so the next pass has less to scan.
+ - Files added after a page is read are new, so they're protected by the 7-day age check (`SubmissionFiles.java:185,189`).
+ - Files for active drafts are always under 7 days old, because a draft expires 7 days after creation.
+ - Queued, importing and needs-reconciliation rows keep their files through the retained set.
+- **The inventory can't grow without limit.** Rows that can't be released are capped per owner: up to 20 active drafts, about 140 imported or failed in the last 7 days, and 1 in needs-reconciliation (the one-active-job rule covers that state). With a few partners, the pass fits well within 10s.
+- **Releasing files after import is safe.** `UploadedFiles.makeMediaAsset` copies each file into the asset store (`copyIn`). Media assets and later derivative generation never read from staging.
+- **Release rule is correct.** Certain failures and imports set `completedAt` (`Submission.java:66,69`). Uncertain failures and stale-claim holds don't set it, and the `completed > 0` check keeps their files.
+- **Daily quota is correct.** It runs under the per-owner lock, after the idempotent-replay check, and counts cancelled drafts (`SubmissionStore.java:43-48`). The test covers both churn rejection and replay after the quota is reached.
+- **Replay batching** filters on `id > :after` in `id` order. Removing a failed row no longer causes other rows to be skipped.
+- **Sweeper budget:** 5000 directories an hour is far more than admission allows (20 drafts × 200 files a day per owner). Failures are isolated per directory.
+
+## Minor issues (none block release)
+
+1. **Operator reconciliation and `completedAt`:** If an operator moves a `needs_reconciliation` row to `imported` or `failed` directly, without setting `completedAt`, its files are kept forever. I couldn't find this covered in the design or plan docs. If the runbook doesn't already say so, add "set `completed_at`" to the manual reconciliation steps.
+2. **Manifest after release:** For an imported or failed draft older than 7 days, `manifest()` now returns an empty file list with no error. It returns 410 for expired or cancelled drafts, so consider doing the same here, or marking the manifest as released.
+3. **One bad row stops the whole pass:** An exception inside `cleanupReference` (for example a commit returning 503) ends that hour's inventory, while the directory sweeper isolates failures per directory. This fails safe (nothing is deleted) and a persistently failing single row is unlikely, but a per-row try/catch would match the sweeper.
+4. **Test gaps:** There's no DB test for the certain-`failed` release path, for more than 100 rows across pages or the deadline pause, or for replay advancing past its first batch. `replayIsBounded` only checks the first batch of 5.
+5. **Shared staging root:** Cleanup builds its retained set from one context only. That's fine while only `context0` exists, but the runbook should say the staging directory must not be shared across contexts.
+6. **Schema:** `completedAt` is a non-null primitive column. That's safe only because the table isn't deployed yet; any pre-existing pilot `SUBMISSION` table should be dropped rather than upgraded in place.
+
+These findings rest on reading the code. They still need your clean build and Java test results to confirm them.
diff --git a/docs/design/submissions/reviews/stage-6-disposition.md b/docs/design/submissions/reviews/stage-6-disposition.md
new file mode 100644
index 0000000000..2d1bbd26d3
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-6-disposition.md
@@ -0,0 +1,9 @@
+# Stage 6 review disposition
+
+Three read-only Claude rounds reviewed the client, contract, integration and pilot handoff. Round 3 found no remaining Critical or Major issues.
+
+Corrections include fresh private configuration reads, complete accepted-response/limit schemas, two resource slots, bounded replay, durable client intent and response reconciliation, row correction/cancellation, numeric JSON normalization, and command-flow tests. The invalid-validation commit response now matches the contract.
+
+Added Java checks for failed-validation commit rejection and replay pagination after round 3. Clarified lost-create recovery before cancellation in the runbook. Conservative uncertain execution and unacknowledged asynchronous indexing remain explicit limitations; operators must reconcile them.
+
+Claude performed source review only. Local runtime fixtures are checked against both schemas. Final executed test results are recorded in ../README.md. QA flag-toggle, browser and deployment gates remain in ../pilot-runbook.md.
diff --git a/docs/design/submissions/reviews/stage-6-round-1.md b/docs/design/submissions/reviews/stage-6-round-1.md
new file mode 100644
index 0000000000..971a7b15a4
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-6-round-1.md
@@ -0,0 +1,90 @@
+I found no Critical issues: nothing lets an import run twice, loses a commit, or leaks the token. There are 7 Major and about 12 Minor issues. The most important is that the on/off flags the runbook relies on for rollback may not take effect at runtime.
+
+One caution: `SubmissionJobs.java` changed while I was reviewing (`postprocess` went from `void` to `boolean`, which fixed a compile error in `SubmissionWorker.java:37`). Findings are against the version I read last. I ran nothing and edited nothing.
+
+## Critical
+None.
+
+## Major
+
+**M1. The on/off flags are cached, so runtime toggling is unreliable.**
+- `submissions.enabled`, `commitEnabled` and `workerEnabled` are read through `CommonConfiguration.getProperty(name, context)`. That keeps a per-JVM cache (`CommonConfiguration.java:46-52`).
+- The cache is only refreshed when some request-scoped config call happens (`:57-62`). Those calls also load per-user/org override files into the same shared entry.
+- So an API-only or worker-only instance may keep old flag values until restart. The runbook says `workerEnabled` "controls processing at runtime", "Admission can be disabled while accepted jobs drain", and its rollback steps depend on these flags.
+- Enrollment is not affected: `getApiAccessProperty` goes through `ShepherdProperties`, which has no cache, so removing a partner takes effect immediately.
+- Fix: read the three flags without the cache, or document that a restart is required. Add a QA gate that toggles each flag on a non-browser instance.
+
+**M2. Contract and runtime disagree, and the contract check can't see it.**
+- The commit response: both specs require `acceptedRevision`, but `SubmissionJobs.java:48` sends `revision`. `examples.json:63` uses `acceptedRevision`, so the check passes anyway.
+- Capabilities `limits`: the design `openapi.yaml` sets `additionalProperties: false` and omits `maxFieldsPerRow` and `maxDraftsPerUser`, which the runtime always sends (`Submissions.java:102-103`). So every real capabilities response fails the design schema. The main `src/main/resources/openapi.yaml:257-264` does include them, so the two specs have diverged.
+- `check_contract.py` only checks the design spec against hand-written examples. It never looks at runtime output or the main spec, so "11 operations/18 examples passed" says nothing about runtime conformance.
+
+**M3. Imports block all uploads and validation on that JVM, and the client gives up after about 7 seconds.**
+- `SubmissionWorker.tick` holds the single intake slot for the whole tick, including the import and generating image derivatives (`SubmissionWorker.java:22`, `SubmissionResources.java:5`).
+- Every upload and validate on that JVM gets 429 meanwhile. The client retries 4 times over about 7 seconds and ignores `Retry-After`.
+- The runbook mentions "one expensive intake operation per JVM" but not that worker imports use that slot. Rerunning does converge, but pilot users will hit this.
+
+**M4. Post-import processing picks the same submissions forever.**
+- `pendingPostprocessing` selects `derivatives == 'complete'` with no check on the indexing phase (`SubmissionJobs.java:129`).
+- So every imported submission is re-sent for indexing on every restart, by every worker JVM.
+- The list is capped at 10,000 sorted oldest first, and nothing is ever deleted. After 10,000 imports, newer submissions never get derivatives or indexing.
+- If indexing is unavailable, the exception ends the loop every tick and blocks everything behind it. The runbook's "replayed idempotently" is true but leaves all this out.
+
+**M5. Fixing rows or a stuck commit leaves orphan drafts and hides the error.**
+- The runbook says to fix validation errors and repeat. But any change to `rows.json` fails the state digest check, so a new state file is needed, which creates a new draft.
+- Old drafts stay live for 7 days and count toward the 20-draft limit. The client has no cancel, and photos are re-uploaded each time.
+- A failed PUT replaces the real 400/413/422 reason with "Rows update unresolved or conflicted" (`client.py:157-160`, `from None`).
+- A saved commit intent that can never succeed (412, `VALIDATION_STALE`, expired) has no documented way out. Starting a new draft is safe while the state is still `draft`/`validated`, but the runbook should say so and cover cancelling the old one.
+
+**M6. The client tests don't exercise recovery.**
+- None of the 4 tests runs `main()`, so they cover none of: create-key persistence, commit intent saved before POST, resume after `queued`, row conflict, pagination, redirect refusal, or the lock.
+- `test_safe_retries_keep_the_original_commit_intent` is circular: it retries a closure over a fixed dict, so it can't fail.
+- The code for these paths looks right, but there is no test evidence for it.
+
+**M7. The runbook overstates verification.**
+- `pilot-runbook.md:4-5` calls README.md "the final verification record".
+- README says "Implementation verification in progress" and that the stage 5/6 reviews are open (README:30, :68). README:47 still shows the old "10 operations / seven examples" line.
+
+## Minor
+- **Upload race:** if a timed-out upload commits between the manifest GET and the retry, the retry gets 412 and aborts instead of re-reading the manifest. A rerun fixes it.
+- **Commit and config staleness:** the contract promises 409 `VALIDATION_STALE` at commit when configuration changed. The runtime only checks this during the import, so the job ends `failed`/`IMPORT_FAILED`.
+- **Wrong or missing status codes:** an expired commit returns 409 `INVALID_STATE`, not 410. Results never return 410.
+- **Retention and "404 after purge":** the capabilities `idempotencyRetentionSeconds` advertises 7 days, but the runtime keeps records forever, so "404 after final purge" never happens.
+- **Schema field name:** the Submission schema says `error`; the runtime sends `errors[]`.
+- **Missing context path:** `statusUrl` and `links.importTask` lack the context path that `Location` includes.
+- **Commit replay needs admission:** the filter checks admission on every non-GET (`SubmissionAuthenticationFilter.java:62-65`). With admission off, a same-key commit replay gets 503. The reference client is fine because it polls, but the contract should say to recover by GET.
+- **Too many jobs marked uncertain:**
+ - Certain-outcome failures (`IllegalStateException("Required media unavailable")`, errors from `shutdownNow` interrupting an import) become `needs_reconciliation`, which blocks the owner.
+ - The runbook has no "drain before restart" step, e.g. confirm no `importing` or `derivatives=running` rows. M1 undermines this anyway.
+- **Indexing never sent after `derivatives=unknown`:** the indexing phase stays `pending` forever. The runbook should say this.
+- **Client input errors crash:** `sorted(names)` runs before the type check, so mixed-type media values raise an uncaught `TypeError`. Malformed `rows.json` raises `KeyError` tracebacks.
+- **Proxy leak on localhost:** for `http://localhost`, a set `http_proxy` with no `no_proxy` sends the bearer token to the proxy in cleartext. Consider an empty `ProxyHandler`.
+- **Client ignores capabilities:** it doesn't check `admissionEnabled`/`commitEnabled`/`stagingAvailable` before uploading.
+- **State durability and lock scope:**
+ - `save()` fsyncs the file but not the directory after `os.replace`.
+ - `flock` may not work across hosts on network filesystems or WSL `/mnt/c`. The runbook should say "local filesystem".
+
+## Checked and correct
+- **Create key:** saved before the POST, and the replay is checked before the draft limit.
+- **Commit intent:** validation ID, key and revision are fsynced before the POST. A same-key replay is checked before `If-Match`. On restart, the POST is skipped once the state is past `validated`.
+- **Upload reconciliation:** by name and SHA-256 from the manifest.
+- **Polling and pagination:** fixed-offset pages over results that never change.
+- **Transport and credentials:**
+ - Redirects are refused; a 3xx surfaces as an `ApiError`.
+ - HTTPS is required outside localhost.
+ - The token comes only from the environment and is never written to state.
+ - State and lock files are 0600, and the lock is taken before state is read.
+- **Worker:**
+ - Stale claims are safely held off and never rerun.
+ - "Imported wins" holds, with a single writer.
+ - Removing a partner fails their unstarted imports as `failed`.
+ - Indexing `unknown` is described as "dispatched", matching your note.
+
+## Convergence
+The client and server converge on resume. The blockers before stage 5/6 sign-off:
+- **M1:** needs either a code fix or a restart requirement in the runbook.
+- **M2:** fix the runtime or both specs, and extend the checker to cover runtime output.
+- **M4:** filter selection on `phase` or add a done marker.
+- **M5 and M6:** the runbook workflow and error surfacing, plus tests that run `main()`.
+
+The rest can be tracked as follow-ups.
diff --git a/docs/design/submissions/reviews/stage-6-round-2.md b/docs/design/submissions/reviews/stage-6-round-2.md
new file mode 100644
index 0000000000..ce01ab027a
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-6-round-2.md
@@ -0,0 +1,66 @@
+I found no Critical issues and one new Major one. The client's check that the server still holds its saved rows gives false conflicts when a numeric field is a whole-number decimal like `2026.0`, and that blocks the runbook's normal validate-then-commit flow. The seven round-1 Majors are resolved. A few smaller contract and evidence gaps remain.
+
+I ran nothing and edited nothing. Everything below comes from reading the code.
+
+## Critical
+None.
+
+## Major
+
+**N1. Client and server normalize numbers differently, so the saved-rows check fails.**
+- The client compares a hash of Python's `json.dumps(..., sort_keys=True)` output (`client.py:124-125`), and it hashes the server's stored rows the same way (`client.py:188-189`).
+- The server stores rows through `SubmissionJson.canonical` → `JSONObject.valueToString` (`SubmissionJson.java:63`). That formatting drops trailing decimal zeros, so `2026.0` is stored and returned as `2026`, and `-8.0` as `-8`.
+- Python reads the server's copy back as the integer `2026`. That hashes differently from the local `2026.0`.
+- **Where it breaks:**
+ - **Rerun with `--commit`:** the first run's PUT succeeds and saves the local hash (`:203`). On the rerun, the server's rows match neither the current input nor the saved hash, so `:190` stops with "Draft rows were changed by another client". This is the flow `pilot-runbook.md:67-70` tells users to follow.
+ - **Row correction:** the same false conflict occurs.
+ - **Lost PUT response:** reconciliation at `:197` and `:201` never matches, so it raises or reports "unresolved".
+- This is likely in practice. Pandas and spreadsheet exports turn integer columns with gaps into floats (`2026.0`), and coordinates like `12.0` are common.
+- Nothing is lost or imported twice, but the client can't finish, and the runbook gives no workaround.
+- The design spec already says clients should reconcile "by GET rows and JSON value equality, not by reproducing this digest" (`openapi.yaml:1189-1190`).
+- **Fix:**
+ - After a PUT succeeds or is reconciled, save the hash of the rows the server returns, not the local hash.
+ - Compare local rows to server rows after normalizing numbers: integral floats become ints, and ideally use `parse_float=Decimal`.
+ - Plain Python `==` won't do, because `True == 1`.
+ - Add a `main()` test with `"Encounter.year": 2026.0`. The current tests use only integers.
+
+## Minor
+
+**Contract and runtime mismatches:**
+- **Commit after an invalid validation:** both specs say it returns 422 `VALIDATION_INVALID` (design `openapi.yaml:794`, published `openapi.yaml:2810-2811`). The runtime returns 409 `VALIDATION_STALE`, because a `valid=false` result leaves the state as `draft` (`Submission.java:48`, `SubmissionJobs.java:34-36`). The reference client never commits an invalid draft, but other clients will get a different error than documented.
+- **Undocumented statuses:** the runtime returns 405 and 408 with code `BAD_REQUEST` (`SubmissionAuthenticationFilter.java:40`, `SubmissionFiles.java:112`). 408 isn't declared anywhere. `SubmissionJson.java:95` pairs 422 with `CAPABILITY_UNAVAILABLE`, while the error description implies 503.
+- **Carried over from round 1, unchanged:**
+ - An expired commit returns 409 `INVALID_STATE`, not 410.
+ - Failures with a known outcome (for example "Required media unavailable" at `SubmissionImporter.java:48`) still become `needs_reconciliation`.
+ - When derivatives end up `unknown`, indexing stays `pending`. This one is now documented (`pilot-runbook.md:121-122`).
+
+**What `check_contract.py --runtime` does and doesn't show:**
+- It checks only two responses (`:95`): capabilities and accepted. Submission, Results, Validation, Manifest, Error and StoredRows are never checked against runtime output. That's how the 422/409 mismatch got through.
+- The accepted file is written from `SubmissionJobs.enqueue` in `SubmissionStoreDbTest:177`, not from the servlet, so the context-path rewrite, `Location` and `ETag` aren't covered.
+- The capabilities file comes from a servlet with no `ServletContext`, so it only exercises `stagingAvailable=false` (`SubmissionsTest:50-59`).
+- It doesn't check that the files in `target/` are fresh.
+- **Evidence status:** the README records no `--runtime` run. Its combined Maven command (`README:78`) doesn't include `SubmissionsTest`, which writes the capabilities file, or `SubmissionPolicyTest`. So runtime conformance and the policy test are claimed but not recorded as passing. Treat them as unverified.
+
+**Worker:**
+- **Replay cost grows with history.** `replayBatch` re-sends indexing for every imported submission with `phase == 'unknown'` on each worker restart, per JVM, 5 per 10-second tick (`SubmissionJobs.java:140-150`). Because `phase` stays `unknown`, this set never shrinks. It terminates and is harmless, but it is O(history) on every restart.
+- **Replay can skip items.** Offset paging skips one item whenever a replayed item switches to `failed` during the scan.
+- **Slot contention.** The worker keeps its slot for the whole tick: the import plus up to 10 derivative generations (`SubmissionWorker.java:25-43`). Concurrent partners on that JVM share the one remaining slot, and a long derivative pass can outlast the client's 5-minute upload budget. A rerun converges, and the runbook (`:35-37`) warns about 429s.
+
+**Client and tests:**
+- **Only 4 tests run as a script.** `test_client.py:60-61` calls `unittest.main()` before `MainFlowTests` is defined, so `python3 test_client.py` runs just 4 tests. `unittest discover`, as in the README, runs all 7. Move the call to the end of the file.
+- **Leftover circular test.** `test_safe_retries_keep_the_original_commit_intent` can't fail. The `main()` tests now cover that path, so delete it.
+- **Create key covered in-process only.** The create-key test covers a retry within one process (the saved key is asserted before each POST). It doesn't cover a process exit after a lost create. The logic is the same, so this is low risk.
+- **Cancel after a lost create.** If the create response is lost, `--cancel` reports "No saved submission to cancel" (`client.py:134-135`) and the server-side draft lingers for 7 days. A normal rerun replays the create key and recovers the ID, after which cancel works. The runbook should say so.
+- **Policy test scope.** `SubmissionPolicyTest` mocks `getApiAccessProperty`. It proves the flags are re-read on each check, not that the file read itself is uncached. The code does confirm that `apiAccessPropsCache` is only filled by tests (`CommonConfiguration.java:39-40, 615-625`). The runbook's QA gate, toggling the flags on an API-only instance (`:172-173`), is still the real proof.
+
+## Round-1 Majors resolved
+- **M1 (flags cached):** the flags and staging path now use `getApiAccessProperty`, which reads the file fresh in production (`SubmissionPolicy.java:13-21`, `SubmissionFiles.java:30`). The runbook is updated and has a QA gate.
+- **M2 (contract vs runtime):** the runtime sends `acceptedRevision` (`SubmissionJobs.java:52`). Both specs list `maxFieldsPerRow` and `maxDraftsPerUser`, and the two specs' `Capabilities` and `Accepted` schemas match. See N1 and the checker notes above for what is still uncovered.
+- **M3 (imports block intake):** there are now two slots, a 429 carries `Retry-After: 5`, and the client honors it within a 5-minute budget.
+- **M4 (post-processing repeats forever):** pending work is selected by `derivatives == 'pending' || (complete && phase == 'pending')`, 10 at a time. Failures are held as `failed`, and the derivative timestamp is separate.
+- **M5 (orphan drafts, hidden errors):** rows can be corrected in the same draft, 4xx details are kept, and `--cancel` exists. `--reset-commit` is allowed only when the server shows `draft` or `validated` with no `operationId`, which `Submission.json` always includes once accepted. Cancel is idempotent (`SubmissionStore.java:84`), and validate doesn't bump the revision, so retrying either is safe.
+- **M6 (tests skip recovery):** the `main()` tests cover a lost create, a lost commit response recovered by GET, resume without a second commit, pagination, row correction, an actionable 422 and the file lock.
+- **M7 (runbook overclaims):** the "final verification record" wording is gone. The README says the full build and frontend regression are underway and that the stage 5/6 reviews are open. It claims no deployment.
+
+## Convergence
+Before stage-6 sign-off, fix N1 and add the float test. Also either align commit-after-invalid with the spec or change the spec. Record the `--runtime` run and the full test command, including `SubmissionsTest` and `SubmissionPolicyTest`, in the README when they actually pass. The full build, frontend regression and the QA gates are still unfinished, so none of them is a pass yet.
diff --git a/docs/design/submissions/reviews/stage-6-round-3.md b/docs/design/submissions/reviews/stage-6-round-3.md
new file mode 100644
index 0000000000..758936540d
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-6-round-3.md
@@ -0,0 +1,48 @@
+**Stage 6, round 3:** I found no Critical or Major issues, so stage 6 has converged. What remains are small gaps in tests and records, plus build evidence you've already said is pending. I didn't count that pending evidence as a false claim.
+
+I only read files. I ran nothing and edited nothing, so I haven't reproduced the seven passing tests or the frontend counts myself; those numbers are as you reported them.
+
+## N1 (rows falsely reported as changed) is fixed
+- **Client side:** `normalize_numbers` (`scripts/submissions/client.py:124-133`) turns whole-number floats into ints, rejects non-finite numbers and keeps booleans separate from ints. It runs before hashing, so the local rows, the saved hash and the server's rows are all hashed the same way.
+- **After an edit:** the client re-reads the rows from the server, checks them against the local rows, and saves the server's hash (`:215-220`). A lost PUT response is also handled correctly (`:208-214`).
+- **Checked against the server's JSON library (org.json 20240303, from `pom.xml`):**
+ - Decimals are stored as `BigDecimal`, and `valueToString` drops trailing zeros. `2026.0` comes back as `2026`, which matches the client.
+ - Exponent forms like `1E+20` come back as a float, which the client then turns into the same int.
+ - `-0.0` comes back as `-0`, which Python reads as `0`, the same as the client's value.
+ - Booleans are stored as `true`/`false` and never turn into `1`/`0`.
+ - The server only prevalidates text and doesn't otherwise change rows (`SubmissionJson.java:16-40, 99-119`).
+
+ I found no remaining number format that would cause a false conflict.
+- **Test:** `test_client.py:104-118` covers the runbook flow end to end:
+ - validate `2026.0` while the fake server stores `2026`;
+ - rerun with `--commit`, without a false conflict and without a second PUT;
+ - recover a lost commit response;
+ - resume without a second commit.
+
+ All seven tests run as a script now that `unittest.main` is at the end of the file (`:177-178`). The old circular test has been replaced by the number/boolean test (`:33-36`).
+
+## Other round-2 items that are now resolved
+- **Commit after an invalid validation:** it returns 422 `VALIDATION_INVALID` when the `validationId` matches the saved `valid=false` report (`SubmissionJobs.java:34-38`). This happens after the revision check and before the stale-validation check, which matches `openapi.yaml:2816-2817`.
+- **Replay skipping items:** `replayBatch` now pages by ID instead of by offset (`SubmissionJobs.java:145-155`, `SubmissionWorker.java:44-53`). Items that switch to `failed` can no longer shift the next page. An interrupted batch is simply repeated, which does no harm.
+- **408:** it is declared for uploads in both specs.
+- **410:** the one remaining 410 (GET files) is correct. The runtime does return 410 `GONE` there (`SubmissionStore.java:106-108`), and `GONE` is in the error-code list.
+
+## Remaining Minor issues (none block sign-off)
+1. **No Java test for the new 422 path.** No test commits against a `valid=false` validation (a search for `VALIDATION_INVALID` in `src/test` finds only the file-upload test). I suggest adding one to `SubmissionStoreDbTest`, and one checking that a stale ID for an invalid validation still returns 409.
+2. **The paged replay is only half tested.** `pendingWorkDoesNotCompeteWithCompletedHistoryAndReplayIsBounded` sets up 7 items but only checks the first page of 5 (`SubmissionStoreDbTest.java:285-286`). Checking that the second page, `replayBatch(..., lastIdOfPage1)`, returns exactly the other 2 would prove the paging works. Replay still re-scans all past submissions on every restart; that's known and harmless.
+3. **The float test only checks the client against itself.** The fake server copies the Java behaviour by assumption (`test_client.py:85`). A Java test that sends `2026.0` and `-0.0` through `replaceRows` and reads the rows back would pin the server side. So would adding stored rows to the `--runtime` fixtures. `check_contract.py --runtime` still checks only capabilities and accepted (`:95`).
+4. **Status and code details:**
+ - 405 and 408 return code `BAD_REQUEST` (`SubmissionAuthenticationFilter.java:40`, `SubmissionFiles.java:112`).
+ - 408 and 405 declare no error body in the specs (`openapi.yaml:2685-2688`), though the runtime sends one.
+ - `SubmissionJson.java:95` still pairs 422 with `CAPABILITY_UNAVAILABLE`.
+5. **README is behind the evidence.**
+ - `README.md:85` still says the frontend regression is "underway". It should record the 21 bulk-import suites passing, and the whole-frontend result (130 passed, 16 failed) with a note that the failing suites are in sources this branch doesn't touch. Calling them "pre-existing" would ideally need a run on the base commit `24cc99aede`.
+ - The combined Maven command (`:78`) still leaves out `SubmissionsTest` and `SubmissionPolicyTest`. That's fine while the full build is pending, but add them when it's recorded.
+6. **Carried over from round 2, still open:**
+ - The runbook doesn't say that `--cancel` after a lost create needs a normal rerun first to recover the submission ID (`pilot-runbook.md:77-78`, `client.py:146-147`).
+ - Failures with a known outcome still become `needs_reconciliation`.
+
+## Remaining gates
+The full Java build that writes the runtime fixtures, the `check_contract.py --runtime` run, the runbook's QA gates (toggling the flags on an API-only instance, and the browser tests) and deployment are all still unfinished. The README correctly doesn't claim them. I'd treat stage 6 as signed off pending those gates. Items 1–2 are cheap and worth doing before the full build, so their results are included in it.
+
+I couldn't save this review because the session is read-only. If you want it on file, it would go in `docs/design/submissions/reviews/stage-6-round-3.md`.
diff --git a/docs/design/submissions/reviews/stage-7-agent-skill-disposition.md b/docs/design/submissions/reviews/stage-7-agent-skill-disposition.md
new file mode 100644
index 0000000000..e376609c67
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-7-agent-skill-disposition.md
@@ -0,0 +1,22 @@
+# Final agent skill review disposition
+
+Added last at the user's request, after the API's full Java regression/build.
+Two actual Claude CLI read-only rounds reviewed the public skill, discovery links,
+registration, examples and recovery instructions against runtime source. Round 2
+found no Critical or Major issues and concluded the skill is ready to publish.
+
+Round-one Majors were corrected: ordinary search tokens produce 401 (wrong
+JWT audience), and a token must be minted with the intended record-owning account.
+The skill requires checking effectiveOwnerId before commit and recommends a
+non-admin integration account. Other corrections cover origin-relative URLs,
+concurrency, HTTP errors, lost validation, configuration rechecks, integer formats,
+subspecies names, response content types and credential-safe curl headers.
+
+Remaining optional nits: indexing can be pending as well as the discussed states;
+an identical upload retry still needs the current ETag. Existing instructions to
+check the manifest/current revision and report unfinished phases cover both.
+
+AgentSkillTest and AgentSkillContentTest: 15 passed, zero failures/errors/skips.
+The suite exercises public serving, links/catalog coverage and runtime parsing of
+the new create/rows examples. Claude performed source review, not test execution.
+Deployment/QA remain separate gates. Final packaging evidence is in ../README.md.
diff --git a/docs/design/submissions/reviews/stage-7-agent-skill-round-1.md b/docs/design/submissions/reviews/stage-7-agent-skill-round-1.md
new file mode 100644
index 0000000000..ae19654050
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-7-agent-skill-round-1.md
@@ -0,0 +1,53 @@
+I found no Critical issues. Two Major issues should be fixed before publishing: the 401/403 guidance for the wrong token type, and not saying whose credentials the token must be minted with. Once those are fixed, the skill is safe and useful for a coding agent to follow. I checked the other claims against the runtime code and found only Minor inaccuracies and gaps. Neither fix requires a runtime change. This was a read-only review; I didn't run any tests or builds.
+
+## Major
+
+**M1. A normal API Access token gets HTTP 401, but the skill says 403.**
+- `submit-sightings.md:333` says a wrong scope or a normal search token gives 403, and `:332` says 401 means "obtain a renewed submissions token".
+- In the code, `verifySubmission` requires the `/submissions` audience (`JwtService.java:132-133`). An ordinary token fails that check and gets **401** "Invalid token" (`SubmissionAuthenticationFilter.java:53-54`).
+- 403 only happens with a valid submissions token: a read-scope token used for a write (`:63`), or an account that isn't enrolled (`SubmissionPolicy.java:29`).
+- **Effect:** an agent given the wrong token will follow the 401 row and ask for a "renewed" token. The person may just create another API Access token, so the agent loops without ever being told the real problem.
+- **Fix:** have the 401 row say that a 401 on the very first request usually means it isn't a `submissions:*` token, so ask for one minted with `scope=submissions:write`. Limit the 403 row to read-scope-on-write and enrollment.
+
+**M2. The skill doesn't say whose credentials the token must be minted with (a record-ownership risk).**
+- `submit-sightings.md:29-31` says "The operator obtains the scoped token … with fresh HTTP Basic credentials".
+- The token's subject is whichever account's Basic credentials were used (`AuthToken.java:52-53,88-89`). Drafts and imported encounters are owned by that account (`SubmissionStore.java:50`, `SubmissionImporter.java:35`).
+- An operator following the text literally could mint the token with their own (possibly admin) account. The records would then be attributed to the operator.
+- An admin token also sees every draft (`SubmissionStore.java:176`), so the claim at `:334` that another owner's submission is "hidden as not found" is only true for non-admin tokens.
+- **Fix:** state that the token must be minted with the credentials of the enrolled account that should own the records. Also say to check `effectiveOwnerId` against that expected account before committing, and to use a non-admin account.
+
+## Minor inaccuracies and gaps
+
+1. **`statusUrl` and `links.importTask` already include the application prefix.** They are site-relative paths like `/wildbook/api/v3/...` (`Submissions.java:62-66`). Adding them to `BASE`, which already ends in `/wildbook`, doubles the prefix. The warning at `:296-297` ("Do not prepend `/api/v3` twice") points at the wrong risk. It should say to resolve these URLs against the server's origin (scheme and host), not against `BASE`.
+2. **Uploads and validation are limited per account, not per draft.** `SubmissionResources.java:5-14` allows one upload/validate per owner across all their drafts, and two across the whole installation. The limit is also per server process, so parallel uploads to different drafts return 429 "Intake processing busy". `:172` should say "sequentially per account". The 429 row (`:342`) should add that the "busy" responses (with `Retry-After: 5`) wrote nothing, so the same request can be retried unchanged.
+3. **Error codes missing from the recovery table.**
+ - 409 `INVALID_STATE`: editing, uploading or validating a frozen or expired draft (`SubmissionStore.java:193`).
+ - 410 `GONE`: `GET /files` on a cancelled or expired draft (`:108`). This can happen during the lost-upload recovery at `:329`.
+ - 400 for a badly formatted `If-Match`, such as unquoted `3` or `W/"3"` (`Submissions.java:93`).
+ - 400 for a rejected filename (`SubmissionFiles.java:64`). The skill states the filename rules but not the error they produce.
+ - 503 `ADMISSION_DISABLED` / `CAPABILITY_UNAVAILABLE` "Commit is disabled". These are definite rejections, not unknown outcomes; the table only covers them with the generic "5xx".
+4. **Enrollment can't be discovered in advance.** `admissionEnabled` is a global setting (`Submissions.java:101`). Whether this particular account is enrolled only shows up as a 403 on the first write, or when the token is minted. Worth one sentence at `:51`.
+5. **No row for a lost validate response.** The safe action is to POST validate again with the same ETag, since revision doesn't change. Each run creates a new report `id`, and only the newest one is accepted at commit (`SubmissionJobs.java:39-41`).
+6. **Indexing-state wording at `:319` is confusing.** The runtime message is "unknown means dispatched; completion is not acknowledged" (`SubmissionJobs.java:126`). The skill says "dispatch was not acknowledged as complete". It also leaves out `indexing.state: "failed"` (`:233`) and `derivatives: "running"/"unknown"` (`:222`).
+7. **Validation isn't pinned to the full configuration.** The config digest covers location IDs, the media-per-encounter limit, the file-size limit and the pixel limit. It does not cover taxonomy, lifeStage or livingStatus (`SubmissionValidator.java:20-24`). A taxonomy change after commit is caught by the re-check at execution time and ends in `failed` (`SubmissionImporter.java:24-28`), not a 409 at commit. So `:276` ("applies only to that exact revision and configuration") overstates what commit-time checking guarantees. It's harmless, but slightly off.
+8. **Numbers such as `2025.0` fail for integer fields.** The integer parser rejects them (`BulkValidator.java:497-503`). `:330` presents 2025 vs 2025.0 only as a comparison quirk. It should also say to send whole integers for year, month, day, hour and minutes.
+9. **Species with three-part names.** `siteTaxonomies` can list names with a subspecies. The check joins `genus + " " + epithet` (`Util.java:403`, `Shepherd.java:2197`), so `specificEpithet` must hold everything after the genus. `:61` and `:132` only describe two-part names.
+10. **A wrong base URL can return an HTML page with HTTP 200, not a 404.** Unmapped paths fall through to the React app (`web.xml:74-77`). The 404 row at `:334` should tell agents to check that responses are `application/json`.
+11. **Curl example.** `--fail-with-body` needs curl 7.76 or later. The token also ends up in the command's arguments. `-H @headerfile` would avoid that, which is better than only warning about it at `:187-188`.
+
+## Confirmed accurate
+
+I checked these claims against the code and they are correct:
+- **Request bodies:** the create body, rows body, field list, envelope 400/422 rules, and strict duplicate-key handling.
+- **Revisions:** the create response replaying revision 0 and `ETag: "0"`, If-Match quoting, rows and uploads incrementing revision while validate does not, and the commit replay matching on key, body and revision.
+- **Field validation:** every row in the failure table (`:250-264`) matches `SubmissionValidator` + `BulkImportUtil` + `BulkValidator`. That includes the error on month when a day is given without one, the leap-year check, the missing-coordinate error landing on the absent field, `INVALID_LOCATION` plus a second `INVALID_VALUE`, and `MISSING_MEDIA` followed by `REQUIRED_VALUE`.
+- **Limits:** 200 rows, 256 fields per row, 2 MiB requests, 200 files, 200 MiB per draft, 24 MP / 16,000 px, 64 KiB multipart overhead, 20 active drafts, 20 new drafts per rolling day, one active job, and 7-day expiry.
+- **Installation settings:** `site-settings` keys (`locationData`, `siteTaxonomies`, `sex`, `lifeStage`, `livingStatus`) and the fact that it needs no token.
+- **Results:** result row shape, cursor paging, the 1–200 limit, and filtering by current access.
+- **Token lifetime:** `expiresInSeconds` and read-scope access surviving removal of write enrollment.
+
+## Registration and tests
+
+- The `AgentSkill.java:36` registration, the `index.md:33-41` entry and the `api-reference.md:11-13` cross-link are correct and consistent. Skills are served without authentication, which is appropriate since they contain no secrets.
+- The new test (`AgentSkillContentTest.java:179-195`) usefully parses both example request bodies through the runtime envelope parsers and checks the field list in both directions.
+- The test doesn't cover the claims most likely to drift: error codes, numeric limits, and the illustrative commit and report JSON. A cheap improvement would assert that the documented codes appear in the runtime sources, and that the documented limits match `SubmissionPolicy` / `SubmissionFiles`.
diff --git a/docs/design/submissions/reviews/stage-7-agent-skill-round-2.md b/docs/design/submissions/reviews/stage-7-agent-skill-round-2.md
new file mode 100644
index 0000000000..f0030888ad
--- /dev/null
+++ b/docs/design/submissions/reviews/stage-7-agent-skill-round-2.md
@@ -0,0 +1,31 @@
+I found no Critical or Major issues in this re-review. Both Majors from round 1 are fixed, and none of the new or changed instructions contradict the runtime code. I think the skill is ready to publish. This was a source-only review with Read/Grep/Glob. I didn't run any tests or builds, so the test and build results you gave are yours, not mine.
+
+## Both Majors are fixed
+
+- **M1 (401 vs 403):** The 401 row (`submit-sightings.md:372`) now says an ordinary token gets 401, even on the first request. That matches the code: `verifySubmission` fails and returns 401 "Invalid token" (`SubmissionAuthenticationFilter.java:50-54`). The 403 row (`:373`) now covers only a read token used for a write (`:63`), an account that isn't enrolled (`SubmissionPolicy.java:29`), or an owner who is no longer eligible at commit (`SubmissionJobs.java:54`). Those are the only 403s an agent can actually get.
+- **M2 (whose account mints the token):** `:29-35` now says to mint with the intended owner's account, recommends a non-admin account, and says to compare `effectiveOwnerId` with the expected user ID. The ID being compared is the right one: it is `user.getId()` throughout, from issuance (`AuthToken.java:81,88-89`) through the allowlist (`SubmissionPolicy.java:23`) and the actor (`SubmissionAuthenticationFilter.java:89`) to the report (`SubmissionValidator.java:79`). The 404 row's "for a non-admin account" qualifier matches `SubmissionStore.java:176`.
+
+## The new instructions match the runtime
+
+- **Returned URLs:** `statusUrl` and `links.importTask` get the application prefix added (`Submissions.java:62-66`), so resolving them against the origin is correct.
+- **Concurrency (`:189-191`):** one expensive operation per owner, two slots per server process, and the worker takes a slot (`SubmissionResources.java:5-11`, `SubmissionWorker.java:25`).
+- **429 "busy" retries:** Both busy responses happen before anything is written and both send `Retry-After: 5` (`SubmissionResources.java:14`, `SubmissionStore.java:208`, `SubmissionAuthenticationFilter.java:94`).
+- **New error rows:**
+ - 400 for a badly formatted `If-Match` matches the pattern check at `Submissions.java:93`.
+ - 409 `INVALID_STATE` (`SubmissionStore.java:193`) and 410 `GONE` (`:107-108`) are accurate.
+ - The 503 codes match: `ADMISSION_DISABLED` (`SubmissionPolicy.java:28`), and "Commit is disabled", which is checked after the idempotent replay (`SubmissionJobs.java:27-32`). So calling these definite rejections is correct.
+- **Enrollment:** `admissionEnabled` is the global flag. Enrollment is checked when a write token is minted and on every write (`AuthToken.java:79-81`, `SubmissionAuthenticationFilter.java:62-65`).
+- **Lost validation response:** Validation doesn't change the revision (`Submission.java:49`), each run creates a new report `id` (`SubmissionValidator.java:74`), and commit accepts only the stored latest report (`SubmissionJobs.java:39-41`).
+- **Indexing and derivative states:** the "unknown" wording matches `SubmissionJobs.java:126`. `failed` is set at `:233`, and `running`/`unknown` at `:162,222`.
+- **Config digest:** It covers only locations, the media-per-encounter limit, the file-size limit and the pixel limit (`SubmissionValidator.java:20-24`). So "execution failure rather than commit-time conflict" is now stated correctly.
+- **Integers and taxonomy:** The integer-format advice matches `BulkValidator.java:494-503`. Putting everything after the genus in `specificEpithet` matches how the server joins the name.
+- **Response shapes:** The rows and file-manifest shapes match `SubmissionStore.java:63,103-104`. An identical same-name upload returns the manifest unchanged (`:131-132`).
+- **OpenAPI:** `/api/v3/docs/openapi.yaml` is mapped (`web.xml:628`), and the spec includes all the submissions paths (`openapi.yaml:2004-2927`).
+- **Curl example:** `-H @-` reads the headers from stdin, and `printf` is a Bash built-in, so the token never appears in the command's arguments. `--fail-with-body` needs curl 7.76 or later, which is newer than when curl added `@file`/`@-` headers, so the stated requirement is enough.
+
+## Optional nits (none block publishing)
+
+1. `indexing.state` can also be `pending` after import (`Submission.java:66`). Before import the `indexing` object is empty, because a null value is dropped. The skill doesn't list `pending`. It's harmless, since agents are already told not to treat anything but a confirmed state as proof.
+2. A lost-upload retry that sends the old ETag gets 412 before the server checks for an identical upload (`SubmissionStore.java:122` runs before `:131`). The skill already says to check `/files` first and then use the current ETag, so the instructions are fine. The sentence at `:234-235` could just say the retry needs the current ETag.
+
+The review has converged. Nothing Critical or Major remains, and none of the amended instructions is inaccurate.
diff --git a/docs/design/submissions/stage-2-operations.md b/docs/design/submissions/stage-2-operations.md
new file mode 100644
index 0000000000..64acd42240
--- /dev/null
+++ b/docs/design/submissions/stage-2-operations.md
@@ -0,0 +1,56 @@
+# Stage 2: private draft API
+
+Historical stage-two increment (see pilot-runbook.md for current configuration).
+This increment implements drafts only: create, inspect, replace/read rows and
+cancel. Upload, validation and commit remain unavailable until their review gates
+pass. Existing bulk import and browser uploads retain their routes and policies.
+Do not enroll production partners until the complete pilot milestone passes.
+
+Configuration (server-side, never supplied by callers):
+
+- Private `apiAccessKeys.properties`: `submissions.enabled=true` enables new writes;
+ omission defaults to disabled.
+- Private `apiAccessKeys.properties`: `submissions.allowedUserIds` is a comma-separated
+ list of enrolled Wildbook user UUIDs. The existing JWT keys and issuer are reused. Submission tokens use the
+ configured audience with `/submissions` appended, distinct from identity tokens. This pilot supports `context0` only.
+
+Fresh HTTP Basic credentials at `POST /api/v3/auth/token?scope=submissions:write`
+issue the explicitly requested write capability only to enrolled users while
+admission is enabled. `scope=submissions:read` lets authenticated owners renew a
+read credential after unenrollment or shutdown. Omitting scope preserves existing
+identity-only tokens, which are rejected by the submissions API.
+
+The new endpoints accept Bearer authentication only. They do not fall back to a
+browser session, inherit its roles or mint a session. Administrators are checked
+against the token's user, not a cookie. Writes also recheck current enrollment.
+
+Create requires an Idempotency-Key; GET returns ETag; PUT rows and DELETE require
+If-Match. A repeated create returns its original body/ETag: GET the resource before
+editing. A cancelled draft stays readable, including its original rows; repeat
+DELETE with the original header is successful. Expired drafts remain readable but
+cannot be edited. No endpoint deletes imported records.
+
+Current bounds: 200 rows, 2 MiB JSON request, 256 fields per row, 20 active drafts
+per account, seven-day draft lifetime. These are published in capabilities.
+Tombstones and operation keys are retained for at least seven days; this increment
+has no physical cleanup job. Later cleanup must preserve that guarantee.
+
+The new `SUBMISSION` table has a UUID primary key, unique scoped create-key hash,
+and a JDO version column. Payloads are bounded JSON text. No legacy table is
+changed. PostgreSQL transaction advisory locks serialize creation per owner and
+mutation per submission across application processes; JDO versioning provides an
+additional check. Every read uses a fresh transaction and refreshes cached values.
+Only a confirmed commit is reported as successful. An uncertain commit returns
+503; clients reconcile or retry their original create key rather than invent one.
+
+Deployment must include DataNucleus enhancement and creation of the new table and
+unique/version constraints. Test against an isolated PostgreSQL before enabling.
+Disable admission to roll back functionality; retain the table and status access.
+
+Stage-two review and test evidence are recorded in `reviews/` and the workbench
+README. This document does not claim later stages are implemented.
+
+Submission tokens are rejected by legacy search/media token verification. Draft
+capacity errors return 429 without suggesting a five-second retry; cancel a draft
+or wait for expiry. JSON parsing rejects invalid UTF-8, duplicate keys, trailing
+content and nesting beyond 32 levels.
diff --git a/docs/design/submissions/stage-3-operations.md b/docs/design/submissions/stage-3-operations.md
new file mode 100644
index 0000000000..0b5d009842
--- /dev/null
+++ b/docs/design/submissions/stage-3-operations.md
@@ -0,0 +1,36 @@
+# Stage 3: private uploads and strict validation
+
+Historical increment notes. For the complete implementation and current private
+configuration/storage requirements, use [pilot-runbook.md](pilot-runbook.md).
+
+This increment adds files GET/POST and validate POST. Commit remains disabled.
+Configure `submissions.stagingDirectory` to an existing private absolute directory,
+owned by the service account, outside the legacy upload tree and web document root.
+The service rejects overlap with legacy uploads. All application instances must see
+the same storage. Restrict directory access to the service account.
+
+One multipart `file` per request, JPEG/PNG only, up to configured media bytes
+(capped at 200 MiB), 24 million decoded pixels, and 16,000 pixels per dimension.
+Multipart overhead is limited to 64 KiB. Draft limits: 200 files, 200 MiB completed
+bytes. A per-draft PostgreSQL transaction lock reserves the write slot before
+streaming. One bounded temporary file can exist in addition to completed capacity
+while retry content is compared; failures before commit remove it. A failed commit
+acknowledgment retains the file because the transaction may actually have committed.
+Crash orphans await the conservative maintenance workflow in the worker stage.
+
+Filenames must be unchanged by Wildbook's filename cleaner, at most 128 characters,
+ASCII letters/digits/dots/underscores/hyphens and start with a letter or digit.
+Case-only conflicts are rejected on every filesystem. Manifest responses never
+expose private blob paths. Lost-response retries use GET files and the current ETag.
+
+Validation copies the payload and reuses BulkImportUtil/BulkValidator. Configured
+location membership is an additional boundary. Actual staged image bytes and
+hashes are checked without creating media, encounters or IA jobs. Each accepted
+row creates a separate new encounter; this pilot does not accept explicit encounter,
+individual, occurrence, project or owner IDs. Per-row media counts therefore match
+the importer's grouping semantics for this supported subset. Partial dates retain
+their precision. Legacy required fields and defaults are unchanged.
+
+Validation saves a report against the same revision, configuration digest and
+manifest digest. Edits clear the report. The draft owner is the effective owner;
+client fields cannot override it. Execution must revalidate before creating records.
diff --git a/docs/plans/2026-09-23-submissions-api-implementation.md b/docs/plans/2026-09-23-submissions-api-implementation.md
new file mode 100644
index 0000000000..9744f27a51
--- /dev/null
+++ b/docs/plans/2026-09-23-submissions-api-implementation.md
@@ -0,0 +1,328 @@
+# Submissions API implementation plan
+
+Status: all six implementation stages are local, 2026-09-23. Claude reviews have
+converged at every stage with no remaining Critical or Major findings. Final
+verification is recorded in the [workbench](../design/submissions/README.md).
+The API is disabled by default; QA pilot/browser gates and deployment remain outstanding.
+
+Based on the [accepted direction](../design/2026-09-23-submissions-engineer-brief.md)
+and [supporting design](../design/2026-09-23-submissions-api.md), checked against
+Wildbook `24cc99aede`.
+
+## Outcome and first milestone
+
+Deliver `/api/v3/submissions` alongside the existing bulk-import API. Reuse its
+field names, validators, media creation and importer. Require configured location
+and strict validation, default to import-only, and initially enable a few approved
+partner accounts. Existing browser imports retain their behavior.
+
+**First runnable milestone:** an enrolled partner can authenticate, discover
+supported input, create a private draft, upload images, submit JSON rows, and
+receive a validation report with stable source-row references. Drafts survive a
+restart. This milestone creates no biological records and starts no IA jobs.
+
+**Pilot release milestone:** the partner can commit that draft once, survive a
+lost HTTP response without duplicate records, and retrieve record IDs and status.
+Uncertain execution outcomes are held for reconciliation instead of retried.
+
+The first milestone is an internal increment, not completion of the intake project.
+
+## Delivery sequence
+
+| Change | Deliverable | Dependency | Exit gate |
+| --- | --- | --- | --- |
+| 1 | Contract and legacy compatibility fixtures | None | Request/response examples and baseline test results recorded |
+| 2 | Pilot access, owned drafts and revision control | 1 | Authorized draft lifecycle survives restart; concurrent creation deduplicates |
+| 3 | Simple uploads and strict preview validation | 2 | First runnable milestone; no domain writes or IA |
+| 4 | Narrow importer adapter and result mapping | 1, 3 | Existing behavior retained; new execution has explicit transaction/side-effect boundary |
+| 5 | Durable commit, worker and results | 2–4 | Concurrent retries and crash scenarios pass against PostgreSQL |
+| 6 | Reference client, operator runbook and pilot | 5 | End-to-end QA plus legacy browser smoke test passes |
+
+Keep each change separately reviewable. A change can span multiple small PRs;
+do not combine mechanical extraction with new behavior in one opaque diff.
+No effort estimate is committed until change 1 establishes the test/build baseline
+and change 4's lifecycle seam is understood.
+
+## 1. Define the contract and establish compatibility
+
+### Tasks
+
+- Create a draft OpenAPI 3.0.3 document under `docs/design/` for the proposed
+ routes; merge implemented operations into `src/main/resources/openapi.yaml`
+ as they become available. `ApiDocsServlet` serves that resource today. Avoid
+ advertising unimplemented routes as working production endpoints.
+- Specify owner/context, UUIDs, revision/ETag, expiry, source metadata,
+ `clientRowId`, row fields, manifest entries, validation reports, errors, and
+ separate import/indexing/detection/identification states.
+- Use examples for create → upload → rows → validate → commit → results.
+ Include error examples, not just the successful sequence.
+- Describe both existing legacy semantics and intentionally stricter new-API
+ semantics in executable fixtures. Snapshot stable fields, not generated IDs,
+ timestamps, log strings or incidental JSON ordering.
+- Run the existing targeted suites before source changes; record any baseline
+ failures without weakening assertions to obtain a green run.
+
+### Contract choices to implement
+
+| Concern | Concrete rule |
+| --- | --- |
+| Version | Envelope `contractVersion: "1"`; reject unsupported versions |
+| Rows | Nonempty array of `{clientRowId, fields}`; unique row IDs per draft |
+| Processing | `import-only` default; advertise other modes only when implemented and enabled |
+| Strictness | Unknown/unsupported fields and invalid rows block commit; no legacy tolerance parameters exposed |
+| Revisions | GET/mutations return quoted ETag; row/file mutations require `If-Match`; missing precondition `428`, stale `412` |
+| Validation | POST targets a revision and returns `200` with `valid`; a check running successfully is not a valid submission |
+| Commit | Requires validation ID, `If-Match` and `Idempotency-Key`; accepted work returns `202` and `Location` |
+| Idempotency | Same key/input returns original resource/operation; different input `409`; replay recognized before stale-revision rejection |
+| Visibility | Draft owner and explicit administrator access; record visibility remains existing Wildbook policy |
+| Errors | Stable code, readable message, request ID and optional row/field/limit; no internal exception dump |
+| Cancellation | Draft DELETE is retry-safe; queued/importing/imported submissions return `409`; no biological record deletion |
+| Expiry | Retain a tombstone for advertised key retention; owner sees `410` for expired drafts during that period |
+
+Canonical request hashes must include effective options, schema version and
+ordered rows, with deterministic object-key ordering. Specify treatment of omitted
+versus explicit defaults. Commit hashes reference immutable validation and manifest
+content. File digests are server-computed. Publish idempotency retention before
+partners depend on it; never imply perpetual deduplication after records expire.
+
+### Legacy fixtures
+
+Cover object rows versus `fieldNames`/array rows, synonyms, date precision,
+submitter defaults, unknown-field tolerance, missing/corrupt images, grouping
+multiple rows into one encounter, existing individual/occurrence links, foreground
+and background status shapes, skip flags, and indexing/IA handoffs.
+
+Existing starting points are `BulkApiPostTest`, `BulkApiOtherTest`,
+`BulkGeneralTest`, `BulkImagesTest`, and `BulkImporterMissingAssetTest` under
+`src/test/java/org/ecocean/api/bulk/`. Some declared Java packages differ from
+their directories; inspect declarations when adding tests. These are useful unit
+fixtures, not proof of durable transactions or concurrency.
+
+## 2. Add pilot access and owned drafts
+
+### Code boundaries
+
+- New `api/Submissions.java`: HTTP routing, parsing, headers and error conversion.
+- New `api/submission/` services: authorization policy, draft persistence and
+ transitions. Names are proposed; keep HTTP logic separate from transactions.
+- New persistent classes under `org.ecocean.submission`, with matching
+ `src/main/resources/org/ecocean/submission/package.jdo` metadata.
+- Add only the new route/filter mappings to `src/main/webapp/WEB-INF/web.xml`.
+- Reuse `api/auth/JwtService.java` and the identity-resolution pattern in
+ `security/WildbookTokenAuthenticationFilter.java`; keep read-path policy intact.
+
+### Tasks
+
+- Add independent feature controls for new admission and commit, both off by
+ default. Disabling admission must leave authorized status/results available
+ for accepted work. Check the enabled partner accounts on each new mutation.
+- Identify pilot users by stable user UUID and installation context. Removing
+ enrollment prevents new mutations; owners can still inspect existing work.
+- Extend token issuance with an explicit, optional import capability request.
+ Existing issuance without that request keeps its current behavior. Verify
+ credentials and enrollment before signing the capability; a scope claim is
+ never accepted merely because it appeared in a request body.
+- Test token subject/context, expiry, account eligibility, signed capability,
+ no cookie fallback for a bad Bearer token, and mixed-identity requests.
+- Support session authentication only with a tested CSRF check for the new
+ writes. If no suitable existing mechanism is available, implement a scoped
+ one before enabling that path; a token-only internal milestone is acceptable
+ if discovery/docs accurately advertise it.
+- Implement draft creation, GET, rows replacement, cancellation and capabilities.
+ Resolve owner on the server. Hide resource existence from unauthorized callers.
+
+### Persistence requirements
+
+Persist submission identity, owner/context, source, revision, state, payload and
+manifest references/hashes, validation reference, reserved ImportTask ID, execution
+metadata, errors and timestamps. Store large content privately and immutably;
+publish a new database reference only after the file is fully written.
+
+Use a separate operation/idempotency record where useful, with a database unique
+constraint on context, principal, operation and key (or a collision-safe bounded
+key digest). Insert the create operation and draft in one transaction. Use database
+optimistic versioning or locking for state transitions; JVM synchronization alone
+does not protect a multi-process deployment.
+
+Verify new metadata is discovered by Maven enhancement and deployed JDO setup.
+Test schema constraints in PostgreSQL and write a rollout/rollback note for added
+tables. Do not assume automatic schema updates prove uniqueness enforcement.
+
+## 3. Add uploads and validation
+
+### Tasks
+
+- Add streaming, single-file multipart upload and manifest GET. Reuse
+ `UploadPaths` containment/name validation. New drafts use private owned staging;
+ old upload routes and staging conventions stay untouched.
+- Reserve quotas atomically before writing, enforce actual-byte limits during
+ streaming, and release reservations on failure. Include multipart overhead in
+ the request bound. Bound decode dimensions/resource use as well as file bytes.
+- Write a temporary file, verify it, then finalize its manifest entry atomically
+ with revision advancement. Clean up a finalized-but-unreferenced file after a
+ failed database update; never make it visible as a completed upload prematurely.
+- Serialize manifest mutation for the first version. For a lost upload response,
+ clients GET the manifest and reconcile digest/name before retrying at the new
+ revision. Parallel uploads with one stale ETag are not silently accepted.
+- Implement same logical filename/digest retry success; reject different content
+ and collisions introduced by filename cleaning or filesystem case behavior.
+- Add a `SubmissionValidator` that copies JSON input, performs new contract and
+ permission checks, and calls `BulkImportUtil.validateRow`/`BulkValidator`.
+- Use the configured location hierarchy (`LocationID.getLocationIDStructure`,
+ also used by `SiteSettings`) to derive valid IDs. Do not treat a format check
+ alone as configured membership, and do not change shared required fields.
+- Validate actual staged media without creating domain objects. Persist the report
+ against its input revision and relevant config digest. A concurrent edit makes
+ the report stale and prevents a transition to `validated`.
+- Count media after the importer's grouping semantics are applied. Reuse/extract
+ a small grouping helper if necessary; do not implement conflicting grouping
+ rules. If this requires change 4 first, keep validation unavailable until then.
+- Restrict the initial supported field set explicitly. Authorize fields that can
+ link or affect existing individuals, occurrences, projects or owners. Discovery
+ lists the supported subset rather than implying every validator field is enabled.
+
+### First milestone demonstration
+
+Using an enrolled test account and two photographs, create a draft, upload both,
+submit valid rows, and see their source IDs and normalized preview. Then show a
+bad location, unknown field, missing image, corrupt image, stale revision and
+cross-owner request produce actionable errors. Restart the application and show
+the draft survives. Verify domain entity counts and IA dispatch counts unchanged.
+
+## 4. Introduce the importer adapter without changing legacy behavior
+
+Read the full lifecycle before extracting: `BulkImport.doPost`, its task/media/IA
+helpers, `BulkImporter.createImport`, `UploadedFiles.makeMediaAsset`, and ImportTask
+status serialization. Include cache handling, post-commit deep individual reindex,
+media derivatives and matching options in the compatibility checklist.
+
+### Tasks
+
+- Create an execution adapter taking explicit context, user identity, reserved
+ ImportTask ID, immutable input, staged files and processing options.
+- Reuse field validation and `BulkImporter` conversion. Do not call a servlet
+ through fake HTTP requests or make authenticated loopback requests to reuse it.
+- Extract only helpers required by both callers. Avoid moving the entire servlet
+ into a new abstraction or altering `processRow` business semantics.
+- Add an optional result collector mapping each client row to actual encounter,
+ occurrence, individual and media IDs. Capture this when rows resolve entities;
+ do not rely on cache iteration order or zip aggregate arrays to input rows.
+- Add an opt-in way to defer derivative/indexing work until after commit, retaining
+ the legacy default. The current importer invokes
+ `MediaAsset.updateStandardChildrenBackground` before its caller commits.
+- Return durable result metadata and post-commit work intent. Use the worker's
+ own Shepherd and reload entities there; never transfer request-scoped JDO
+ objects into a background thread.
+
+Gate this change on the legacy fixtures plus integration assertions that imported
+records and row mappings match the legacy equivalent, shared entities stay
+consistent, and no new-path side effect runs before a successful commit.
+
+## 5. Add durable commit and results
+
+### Tasks
+
+- Under a database lock/version check, verify authorization, revision, completed
+ files and validation; freeze input, reserve an ImportTask ID, persist `queued`
+ and the operation response, then return `202`. One submission has at most one
+ accepted execution even when callers use different idempotency keys.
+- Recheck current policy/configuration at execution. If input interpretation or
+ authorization changed, fail before domain creation with an explicit reason;
+ never silently apply new defaults to a previously reviewed draft.
+- Use bounded workers with durable claims and per-installation/user concurrency
+ limits. Integrate worker startup/shutdown into existing application lifecycle
+ after locating the appropriate hook; do not add an untracked servlet thread.
+- Persist domain objects, row mappings, imported state and post-commit intent in
+ one transaction where supported. Separate progress transactions are advisory.
+- Handle database commit errors as potentially uncertain until reconciled. A
+ worker lease expiring does not establish rollback; use a fencing mechanism
+ before automatic takeover. Conservative manual reconciliation is sufficient
+ for the first pilot when the outcome cannot be established safely.
+- Execute/reconcile derivatives, indexing and requested IA from persisted intent.
+ Preserve imported state if downstream work fails. Advertise only processing
+ modes whose dispatch/recovery behavior is covered; IA uncertainty is explicit.
+- Add paginated results with stable ordering/cursors and source-row mapping.
+ Return links to existing task/record pages, filtered by current authorization.
+- Add expiry and orphan cleanup that excludes active and uncertain jobs, respects
+ key/tombstone retention and never removes shared or pre-existing assets.
+
+### Required failure-injection tests
+
+| Scenario | Expected result |
+| --- | --- |
+| Two concurrent create requests, same key/input | One draft; both resolve to it |
+| Same key, changed input | `409`, no second draft/import |
+| Concurrent commit, same or different keys | One accepted execution |
+| Response lost after queue transaction commits | Retry returns original operation |
+| Crash before worker claims queued job | Job remains durably discoverable |
+| Crash during import transaction | Rollback proven before retry, or reconciliation state |
+| Crash after domain commit before dispatch | Records retained; persisted intent available |
+| Worker lease expires while worker still runs | No second concurrent writer |
+| IA accepts work but dispatch acknowledgement is lost | No blind duplicate dispatch; reconciliation if no deduplication proof |
+| Disk failure or DB rollback after copying assets | No false success; safe orphan cleanup |
+| Feature admission disabled during execution | Accepted work drains/reconciles; status remains readable |
+
+Use real PostgreSQL transactions and independent persistence contexts for
+concurrency/recovery tests. Mock-based servlet tests cannot establish these
+guarantees. Reset process-wide PMF/configuration state between container tests.
+
+## 6. Deliver a usable pilot
+
+- Publish implemented OpenAPI operations, exact limits and retry rules. Keep
+ capabilities consistent with enabled processing and upload modes.
+- Add a small reference client using only the documented HTTP contract. It stores
+ submission/operation IDs before retrying, reconciles uploads, presents validation
+ errors, commits explicit revisions and polls with backoff. Keep credentials out
+ of example source and logs.
+- Document how operators enroll/remove partners, choose limits, disable admission,
+ inspect stuck work, reconcile uncertainty and expire drafts safely.
+- Track accepted/failed/reconciled submissions, time spent in each phase, queue
+ age, upload bytes/quota use and downstream status. Log correlation IDs and
+ counts without raw tokens, full row payloads or sensitive locations.
+- Pilot on QA with one selected partner integration. Compare a representative
+ legacy import and new API import, including grouped rows and partial dates.
+- Confirm no regressions in the browser upload → review → import → task workflow.
+ Broader enrollment follows observed results; deployment is a separate action.
+
+## Verification commands and evidence
+
+These are planned checks, **not reported passes** from this documentation change.
+
+```bash
+mvn test -Dtest=BulkApiPostTest,BulkApiOtherTest,BulkGeneralTest,BulkImagesTest,BulkImporterMissingAssetTest
+mvn test -Dtest=AuthTokenTest,AuthTokenStepUpTest,WildbookTokenAuthenticationFilterTest
+mvn test -Dtest='UploadPaths*Test'
+```
+
+Run new submissions tests at each stage, then `mvn clean install` for the final
+integration gate, including DataNucleus enhancement. Capture command, revision,
+result, environment and baseline failures in each PR. Run frontend regression
+checks using the repository's CI Jest runner from `frontend/`:
+
+```bash
+CI=true npx jest --ci --runInBand --testPathPattern='BulkImport|bulkImport'
+```
+
+Existing frontend test failures must be distinguished from introduced failures.
+Manual browser smoke testing remains necessary for unchanged-client compatibility.
+
+## Rollback and deferred work
+
+Disable new admission/commit, keep status access and drain/reconcile accepted work.
+Retain new tables and private artifacts while work or retention obligations remain.
+An older application rollback requires workers stopped and queued work accounted
+for; do not assume removing the feature flag makes an in-flight import disappear.
+
+Defer resumable uploads, general upserts, anonymous intake, new UI, spreadsheet
+parsing, webhooks and broad delegated OAuth. Universally requiring location in
+legacy bulk import is a separate compatibility change. Keep the agreed ownership,
+strict validation and commit-retry behavior in the pilot scope.
+
+## Final requested addition: agent skill
+
+After the six-stage API implementation and successful full Java build, added the
+public `submit-sightings` skill, registered in the existing AgentSkill catalog and
+linked from the base toolbox and API reference. It documents every supported field,
+wire formats, configured-value discovery, concrete validation failures and safe
+recovery. Fifteen skill tests passed; two Claude review rounds converged with no
+Critical/Major findings. See the workbench for transcripts and packaging evidence.
diff --git a/pom.xml b/pom.xml
index 09b8c844b4..b15cbc0fae 100644
--- a/pom.xml
+++ b/pom.xml
@@ -637,6 +637,11 @@
${jjwt.version}runtime
+
+ com.fasterxml.jackson.core
+ jackson-core
+ 2.17.0
+
diff --git a/scripts/submissions/.gitignore b/scripts/submissions/.gitignore
new file mode 100644
index 0000000000..c18dd8d83c
--- /dev/null
+++ b/scripts/submissions/.gitignore
@@ -0,0 +1 @@
+__pycache__/
diff --git a/scripts/submissions/check_contract.py b/scripts/submissions/check_contract.py
new file mode 100644
index 0000000000..ef97656e60
--- /dev/null
+++ b/scripts/submissions/check_contract.py
@@ -0,0 +1,103 @@
+#!/usr/bin/env python3
+"""Check the draft's references, examples and critical HTTP contract invariants.
+
+Requires PyYAML and jsonschema. This is not a substitute for an OpenAPI validator
+or runtime integration tests.
+"""
+import json
+import datetime
+import uuid
+from pathlib import Path
+
+import jsonschema
+import yaml
+
+
+ROOT = Path(__file__).resolve().parents[2]
+CONTRACT = ROOT / "docs/design/submissions"
+spec = yaml.safe_load((CONTRACT / "openapi.yaml").read_text())
+formats = jsonschema.FormatChecker()
+
+
+@formats.checks("uuid", raises=(ValueError, AttributeError))
+def valid_uuid(value):
+ return not isinstance(value, str) or str(uuid.UUID(value)) == value.lower()
+
+
+@formats.checks("date-time", raises=(ValueError, TypeError))
+def valid_timestamp(value):
+ if not isinstance(value, str):
+ return True
+ return "T" in value and datetime.datetime.fromisoformat(value.replace("Z", "+00:00")).tzinfo is not None
+
+
+def resolve(value):
+ if isinstance(value, dict):
+ if "$ref" in value:
+ target = spec
+ assert value["$ref"].startswith("#/"), value
+ for part in value["$ref"][2:].split("/"):
+ target = target[part.replace("~1", "/").replace("~0", "~")]
+ return resolve(target)
+ return {key: resolve(item) for key, item in value.items()}
+ if isinstance(value, list):
+ return [resolve(item) for item in value]
+ return value
+
+
+expanded = resolve(spec)
+operations = []
+for path, methods in expanded["paths"].items():
+ assert path.startswith("/api/v3/submissions")
+ for method, operation in methods.items():
+ operations.append(operation["operationId"])
+ parameters = operation.get("parameters", [])
+ if "{id}" in path:
+ assert any(p["name"] == "id" and p["required"] for p in parameters)
+ if method in ("post", "put", "delete") and "{id}" in path:
+ assert any(p["name"] == "If-Match" and p["required"] for p in parameters)
+ assert "412" in operation["responses"] and "428" in operation["responses"]
+ assert "401" in operation["responses"]
+ assert "403" in operation["responses"]
+ assert "Retry-After" in operation["responses"]["429"]["headers"]
+ if "requestBody" in operation:
+ for media in operation["requestBody"]["content"].values():
+ jsonschema.Draft4Validator.check_schema(media["schema"])
+assert len(operations) == len(set(operations))
+
+for name, example in json.loads((CONTRACT / "examples.json").read_text()).items():
+ schema = expanded["components"]["schemas"][example["schema"]]
+ errors = list(jsonschema.Draft4Validator(
+ schema, format_checker=formats
+ ).iter_errors(example["value"]))
+ assert bool(errors) != example.get("valid", True), (name, errors)
+ print("Checked example:", name)
+
+create = expanded["components"]["schemas"]["Create"]
+assert create["properties"]["processing"]["properties"]["mode"]["default"] == "import-only"
+commit = expanded["paths"]["/api/v3/submissions/{id}/commit"]["post"]
+assert "202" in commit["responses"]
+assert any(p["name"] == "Idempotency-Key" and p["required"] for p in commit["parameters"])
+assert any(p["name"] == "Idempotency-Key" and p["required"] for p in
+ expanded["paths"]["/api/v3/submissions"]["post"]["parameters"])
+for path, method in [("/api/v3/submissions/{id}", "get"),
+ ("/api/v3/submissions/{id}/rows", "put"),
+ ("/api/v3/submissions/{id}/rows", "get"),
+ ("/api/v3/submissions/{id}/files", "post"),
+ ("/api/v3/submissions/{id}/validate", "post")]:
+ assert "ETag" in expanded["paths"][path][method]["responses"]["200"]["headers"]
+print(f"Checked {len(operations)} operations and all local references.")
+
+
+# Optional runtime evidence, emitted by the servlet and PostgreSQL acceptance tests.
+if "--runtime" in __import__("sys").argv:
+ published = yaml.safe_load((ROOT / "src/main/resources/openapi.yaml").read_text())
+ for label, schema_name in [("capabilities", "Capabilities"), ("accepted", "Accepted")]:
+ value = json.loads((ROOT / "target" / ("submissions-" + label + ".json")).read_text())
+ jsonschema.Draft4Validator(expanded["components"]["schemas"][schema_name], format_checker=formats).validate(value)
+ original = spec
+ spec = published
+ runtime_schema = resolve(published["components"]["schemas"]["SubmissionApi" + schema_name])
+ spec = original
+ jsonschema.Draft4Validator(runtime_schema, format_checker=formats).validate(value)
+ print("Checked runtime response against both specs:", label)
diff --git a/scripts/submissions/client.py b/scripts/submissions/client.py
new file mode 100644
index 0000000000..4c07fd241a
--- /dev/null
+++ b/scripts/submissions/client.py
@@ -0,0 +1,300 @@
+#!/usr/bin/env python3
+"""Resumable pilot client; credentials come only from WILDBOOK_SUBMISSIONS_TOKEN."""
+import argparse
+import hashlib
+import fcntl
+import tempfile
+import json
+import os
+from pathlib import Path
+import time
+import urllib.error
+import urllib.parse
+import urllib.request
+import uuid
+
+
+class ApiError(Exception):
+ def __init__(self, status, body, retry_after=None):
+ self.status, self.body, self.retry_after = status, body, retry_after
+ super().__init__(f"HTTP {status}: {body}")
+
+
+class Client:
+ def __init__(self, base, token):
+ parts = urllib.parse.urlsplit(base)
+ if parts.scheme != "https" and not (parts.scheme == "http" and parts.hostname in {"localhost", "127.0.0.1"}):
+ raise ValueError("Use HTTPS (HTTP is allowed only on localhost)")
+ if parts.username or parts.password or parts.query or parts.fragment:
+ raise ValueError("Base URL must not contain credentials, query or fragment")
+ self.base, self.token = base.rstrip("/"), token
+ # Never forward a bearer credential to a redirect target.
+ class NoRedirect(urllib.request.HTTPRedirectHandler):
+ def redirect_request(self, req, fp, code, msg, headers, newurl):
+ return None
+ self.http = urllib.request.build_opener(urllib.request.ProxyHandler({}), NoRedirect())
+
+ def request(self, method, path, data=None, headers=None):
+ hdr = {"Authorization": "Bearer " + self.token, "Accept": "application/json"}
+ if headers:
+ hdr.update(headers)
+ if isinstance(data, dict):
+ data = json.dumps(data, separators=(",", ":")).encode()
+ hdr["Content-Type"] = "application/json"
+ req = urllib.request.Request(self.base + path, data=data, headers=hdr, method=method)
+ try:
+ with self.http.open(req, timeout=150) as response:
+ body = response.read()
+ return json.loads(body) if body else {}
+ except urllib.error.HTTPError as ex:
+ raw = ex.read(65536).decode("utf-8", errors="replace")
+ raise ApiError(ex.code, raw, ex.headers.get("Retry-After")) from None
+
+
+def save(path, state):
+ fd, temporary = tempfile.mkstemp(prefix=path.name + ".", dir=path.parent)
+ with os.fdopen(fd, "w") as out:
+ json.dump(state, out, indent=2)
+ out.flush()
+ os.fsync(out.fileno())
+ os.replace(temporary, path)
+ directory = os.open(path.parent, os.O_RDONLY | os.O_DIRECTORY)
+ try:
+ os.fsync(directory)
+ finally:
+ os.close(directory)
+
+
+def retry_safe(call, wait_seconds=300):
+ deadline, attempt = time.monotonic() + wait_seconds, 0
+ while True:
+ delay = min(30, 2 ** min(attempt, 5))
+ try:
+ return call()
+ except ApiError as ex:
+ if ex.status not in {429, 500, 502, 503, 504}:
+ raise
+ if ex.retry_after and ex.retry_after.isdigit():
+ delay = min(60, max(1, int(ex.retry_after)))
+ if time.monotonic() + delay >= deadline:
+ raise
+ except (urllib.error.URLError, TimeoutError, ConnectionError):
+ if time.monotonic() + delay >= deadline:
+ raise
+ time.sleep(delay)
+ attempt += 1
+
+
+def upload(client, route, file, max_bytes):
+ if file.stat().st_size > max_bytes:
+ raise ValueError(f"File exceeds installation limit: {file.name}")
+ content = file.read_bytes()
+ digest = hashlib.sha256(content).hexdigest()
+ boundary = "wildbook-" + uuid.uuid4().hex
+ if any(c in file.name for c in '\r\n"\\') or not file.name.isascii():
+ raise ValueError("Use safe ASCII filenames")
+ payload = (f'--{boundary}\r\nContent-Disposition: form-data; name="file"; filename="{file.name}"\r\n'
+ 'Content-Type: application/octet-stream\r\n\r\n').encode() + content + f"\r\n--{boundary}--\r\n".encode()
+ deadline = time.monotonic() + 300
+ while True:
+ manifest = retry_safe(lambda: client.request("GET", route + "/files"))
+ for entry in manifest["files"]:
+ if entry["name"] == file.name:
+ if entry["sha256"] != digest:
+ raise ValueError(f"Different content already uploaded as {file.name}")
+ return
+ delay = 5
+ try:
+ client.request("POST", route + "/files", payload,
+ {"Content-Type": "multipart/form-data; boundary=" + boundary,
+ "If-Match": f'"{manifest["revision"]}"'})
+ return
+ except ApiError as ex:
+ if ex.status not in {412, 429, 500, 502, 503, 504}:
+ raise
+ if ex.retry_after and ex.retry_after.isdigit():
+ delay = min(60, max(1, int(ex.retry_after)))
+ except (urllib.error.URLError, TimeoutError, ConnectionError):
+ pass
+ if time.monotonic() + delay >= deadline:
+ raise RuntimeError("Upload outcome unresolved; rerun with this state file")
+ time.sleep(delay)
+
+
+def normalize_numbers(value):
+ if isinstance(value, dict):
+ return {key: normalize_numbers(item) for key, item in value.items()}
+ if isinstance(value, list):
+ return [normalize_numbers(item) for item in value]
+ if type(value) is float:
+ if not __import__("math").isfinite(value):
+ raise ValueError("JSON numbers must be finite")
+ return int(value) if value.is_integer() else value
+ return value # bool stays distinct from int
+
+
+def digest(value):
+ return hashlib.sha256(json.dumps(normalize_numbers(value), sort_keys=True).encode()).hexdigest()
+
+
+def run(args, client):
+ state = json.loads(args.state.read_text()) if args.state.exists() else None
+ if state and (state["baseUrl"] != client.base or state["source"] != args.source):
+ raise ValueError("State belongs to a different source or installation")
+ root = "/api/v3/submissions"
+ if args.cancel:
+ if not state or "id" not in state:
+ raise ValueError("No saved submission to cancel")
+ route = root + "/" + state["id"]
+ current = retry_safe(lambda: client.request("GET", route))
+ retry_safe(lambda: client.request("DELETE", route, headers={"If-Match": f'"{current["revision"]}"'}))
+ state["cancelled"] = True
+ save(args.state, state)
+ print("Cancelled", state["id"])
+ return 0
+ if not args.rows or not args.media_dir:
+ raise ValueError("--rows and --media-dir are required except for --cancel")
+ rows = json.loads(args.rows.read_text())
+ if not isinstance(rows, dict) or not isinstance(rows.get("rows"), list) or not rows["rows"]:
+ raise ValueError("Input must be an object containing a nonempty rows array")
+ names = set()
+ for row in rows["rows"]:
+ if not isinstance(row, dict) or not isinstance(row.get("fields"), dict) or not isinstance(row.get("clientRowId"), str):
+ raise ValueError("Each row needs clientRowId and a fields object")
+ for key, name in row["fields"].items():
+ if key.startswith("Encounter.mediaAsset"):
+ if not isinstance(name, str) or Path(name).name != name or name in {".", ".."}:
+ raise ValueError("Media references must be filenames")
+ names.add(name)
+ rows_digest = digest(rows)
+ if state is None:
+ state = {"createKey": str(uuid.uuid4()), "rowsDigest": rows_digest, "baseUrl": client.base, "source": args.source}
+ if state.get("cancelled"):
+ raise ValueError("Submission was cancelled; use a new state file for a new batch")
+ save(args.state, state)
+ if "id" not in state:
+ caps = retry_safe(lambda: client.request("GET", root + "/capabilities"))
+ if not caps["admissionEnabled"] or not caps.get("stagingAvailable", False):
+ raise ValueError("New intake is unavailable; retain the state and try later")
+ created = retry_safe(lambda: client.request("POST", root,
+ {"contractVersion": "1", "source": {"name": args.source}}, {"Idempotency-Key": state["createKey"]}))
+ state["id"] = created["id"]
+ save(args.state, state)
+ route = root + "/" + state["id"]
+ if args.reset_commit:
+ current = retry_safe(lambda: client.request("GET", route))
+ if current["state"] not in {"draft", "validated"} or "operationId" in current:
+ raise ValueError("Cannot reset an accepted execution; inspect status/results")
+ for key in ["commitRequest", "commitKey", "commitRevision", "operationId"]:
+ state.pop(key, None)
+ save(args.state, state)
+ if "commitRequest" in state and state["rowsDigest"] != rows_digest:
+ raise ValueError("Commit intent is frozen; inspect status, then use --reset-commit only if it was never accepted")
+ if "commitRequest" not in state:
+ caps = retry_safe(lambda: client.request("GET", root + "/capabilities"))
+ if not caps["admissionEnabled"] or not caps.get("stagingAvailable", False):
+ raise ValueError("Intake is unavailable; retain the state and try later")
+ for name in sorted(names):
+ upload(client, route, args.media_dir / name, caps["limits"]["maxFileBytes"])
+ stored = retry_safe(lambda: client.request("GET", route + "/rows"))
+ if digest({"rows": stored["rows"]}) != rows_digest:
+ if stored["rows"] and digest({"rows": stored["rows"]}) != state["rowsDigest"]:
+ raise ValueError("Draft rows were changed by another client; inspect before replacing")
+ try:
+ client.request("PUT", route + "/rows", rows, {"If-Match": f'"{stored["revision"]}"'})
+ except ApiError as ex:
+ if ex.status < 500 and ex.status != 429:
+ raise # retain actionable validation/conflict response
+ check = retry_safe(lambda: client.request("GET", route + "/rows"))
+ if digest({"rows": check["rows"]}) != rows_digest:
+ raise
+ except (urllib.error.URLError, TimeoutError, ConnectionError):
+ check = retry_safe(lambda: client.request("GET", route + "/rows"))
+ if digest({"rows": check["rows"]}) != rows_digest:
+ raise RuntimeError("Rows update unresolved; rerun with this state file") from None
+ confirmed = retry_safe(lambda: client.request("GET", route + "/rows"))
+ server_digest = digest({"rows": confirmed["rows"]})
+ if server_digest != rows_digest:
+ raise ValueError("Rows changed before validation; inspect the draft")
+ state["rowsDigest"] = server_digest
+ save(args.state, state)
+ current = retry_safe(lambda: client.request("GET", route))
+ report = retry_safe(lambda: client.request("POST", route + "/validate", {}, {"If-Match": f'"{current["revision"]}"'}))
+ print(json.dumps({"submissionId": state["id"], "valid": report["valid"], "errors": report["errors"]}, indent=2))
+ if not report["valid"] or not args.commit:
+ return 0 if report["valid"] else 2
+ if not caps["commitEnabled"]:
+ raise ValueError("Commit is disabled; draft is retained")
+ state["commitRequest"] = {"validationId": report["id"]}
+ state["commitKey"] = str(uuid.uuid4())
+ state["commitRevision"] = report["revision"]
+ save(args.state, state)
+ status = retry_safe(lambda: client.request("GET", route))
+ if status["state"] in {"draft", "validated"}:
+ if not args.commit:
+ raise ValueError("Commit intent is saved; use --commit to resume it")
+ def commit_once():
+ try:
+ return client.request("POST", route + "/commit", state["commitRequest"],
+ {"Idempotency-Key": state["commitKey"], "If-Match": f'"{state["commitRevision"]}"'})
+ except (ApiError, urllib.error.URLError, TimeoutError, ConnectionError):
+ observed = retry_safe(lambda: client.request("GET", route))
+ if "operationId" in observed:
+ return observed
+ raise
+ accepted = retry_safe(commit_once)
+ state["operationId"] = accepted["operationId"]
+ save(args.state, state)
+ deadline, delay = time.monotonic() + args.poll_seconds, 2
+ while True:
+ status = retry_safe(lambda: client.request("GET", route))
+ if status["state"] in {"imported", "failed", "needs_reconciliation", "cancelled", "expired"}:
+ break
+ if time.monotonic() >= deadline:
+ print("Still processing; rerun with the same state file.")
+ return 3
+ time.sleep(delay)
+ delay = min(30, delay * 2)
+ page_path = route + "/results"
+ while True:
+ page = retry_safe(lambda: client.request("GET", page_path))
+ print(json.dumps(page, indent=2))
+ if "nextCursor" not in page:
+ break
+ page_path = route + "/results?cursor=" + urllib.parse.quote(page["nextCursor"])
+ return 0 if status["state"] == "imported" else 2
+
+
+def main():
+ parser = argparse.ArgumentParser(description=__doc__)
+ parser.add_argument("--base-url", required=True)
+ parser.add_argument("--rows", type=Path)
+ parser.add_argument("--media-dir", type=Path)
+ parser.add_argument("--state", type=Path, required=True)
+ parser.add_argument("--source", default="submissions-reference-client")
+ parser.add_argument("--commit", action="store_true")
+ parser.add_argument("--cancel", action="store_true", help="Cancel an editable saved draft")
+ parser.add_argument("--reset-commit", action="store_true", help="Clear unaccepted commit intent after checking server state")
+ parser.add_argument("--poll-seconds", type=int, default=900)
+ args = parser.parse_args()
+ token = os.environ.get("WILDBOOK_SUBMISSIONS_TOKEN")
+ if not token:
+ parser.error("Set WILDBOOK_SUBMISSIONS_TOKEN to an explicitly scoped bearer token")
+ transport = Client(args.base_url, token)
+ lock_fd = os.open(str(args.state) + ".lock", os.O_WRONLY | os.O_CREAT | os.O_NOFOLLOW, 0o600)
+ try:
+ try:
+ fcntl.flock(lock_fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
+ except BlockingIOError:
+ raise ValueError("Another client is using this state file") from None
+ return run(args, transport)
+ finally:
+ os.close(lock_fd)
+
+
+if __name__ == "__main__":
+ try:
+ raise SystemExit(main())
+ except (ApiError, ValueError, RuntimeError, OSError) as error:
+ print(str(error), file=__import__("sys").stderr)
+ raise SystemExit(2)
diff --git a/scripts/submissions/test_client.py b/scripts/submissions/test_client.py
new file mode 100644
index 0000000000..5c833cc43b
--- /dev/null
+++ b/scripts/submissions/test_client.py
@@ -0,0 +1,178 @@
+import importlib.util
+from pathlib import Path
+import tempfile
+import unittest
+from unittest.mock import patch
+import urllib.error
+import hashlib
+
+spec = importlib.util.spec_from_file_location("client", Path(__file__).with_name("client.py"))
+client = importlib.util.module_from_spec(spec)
+spec.loader.exec_module(client)
+
+
+class RecoveryTests(unittest.TestCase):
+ def test_lost_upload_response_reconciles_manifest_without_second_upload(self):
+ class Server:
+ def __init__(self):
+ self.files = []
+ self.uploads = 0
+ def request(self, method, path, data=None, headers=None):
+ if method == "GET":
+ return {"revision": len(self.files), "files": self.files}
+ self.uploads += 1
+ self.files = [{"name": "a.png", "sha256": hashlib.sha256(b"test-image").hexdigest()}]
+ raise urllib.error.URLError("response lost after save")
+ server = Server()
+ with tempfile.TemporaryDirectory() as root, patch.object(client.time, "sleep"):
+ image = Path(root) / "a.png"
+ image.write_bytes(b"test-image")
+ client.upload(server, "/draft", image, 1000)
+ self.assertEqual(1, server.uploads)
+
+ def test_numeric_equality_preserves_boolean_distinction(self):
+ self.assertEqual(client.digest({"year": 2026.0}), client.digest({"year": 2026}))
+ self.assertEqual(client.digest({"latitude": -0.0}), client.digest({"latitude": 0}))
+ self.assertNotEqual(client.digest({"year": True}), client.digest({"year": 1}))
+
+ def test_conflicts_are_never_automatically_retried(self):
+ attempts = []
+ def conflict():
+ attempts.append(1)
+ raise client.ApiError(412, "stale revision")
+ with self.assertRaises(client.ApiError):
+ client.retry_safe(conflict)
+ self.assertEqual(1, len(attempts))
+
+ def test_transport_rejects_credentials_and_remote_cleartext(self):
+ for url in ["http://example.org", "https://user:password@example.org", "https://example.org?token=x"]:
+ with self.assertRaises(ValueError):
+ client.Client(url, "test-token")
+
+
+
+class MainFlowTests(unittest.TestCase):
+ def test_main_persists_keys_reconciles_lost_responses_resumes_and_paginates(self):
+ import json
+ import os
+ import sys
+ class Server:
+ base = "http://localhost"
+ def __init__(self, state):
+ self.state_file = state
+ self.rows = []
+ self.revision = 0
+ self.state = "draft"
+ self.creates = []
+ self.commits = []
+ self.pages = []
+ self.lost_create = False
+ def request(self, method, path, data=None, headers=None):
+ saved = json.loads(self.state_file.read_text())
+ if path.endswith("/capabilities"):
+ return {"admissionEnabled": True, "commitEnabled": True, "stagingAvailable": True, "limits": {"maxFileBytes": 10000}}
+ if method == "POST" and path == "/api/v3/submissions":
+ self.creates.append(headers["Idempotency-Key"])
+ assert saved["createKey"] == headers["Idempotency-Key"]
+ if not self.lost_create:
+ self.lost_create = True
+ raise urllib.error.URLError("create response lost")
+ return {"id": "saved-id"}
+ if path.endswith("/rows"):
+ if method == "PUT":
+ self.rows, self.revision = json.loads(json.dumps(data["rows"])), self.revision + 1
+ for row in self.rows:
+ row["fields"] = {k: int(v) if type(v) is float and v.is_integer() else v for k, v in row["fields"].items()}
+ return {"rows": self.rows, "revision": self.revision}
+ if path.endswith("/validate"):
+ self.state = "validated"
+ return {"id": "validation-id", "revision": self.revision, "valid": True, "errors": []}
+ if path.endswith("/commit"):
+ assert saved["commitRequest"] == data
+ assert saved["commitKey"] == headers["Idempotency-Key"]
+ assert headers["If-Match"] == f'"{saved["commitRevision"]}"'
+ self.commits.append(data.copy())
+ self.state = "imported"
+ raise urllib.error.URLError("accepted response lost")
+ if "/results" in path:
+ self.pages.append(path)
+ return {"rows": []} if "cursor=" in path else {"rows": [], "nextCursor": "1"}
+ result = {"state": self.state, "revision": self.revision}
+ if self.state == "imported":
+ result["operationId"] = "original-operation"
+ return result
+ with tempfile.TemporaryDirectory() as root:
+ state, rows = Path(root) / "state.json", Path(root) / "rows.json"
+ rows.write_text(json.dumps({"rows": [{"clientRowId": "one", "fields": {"Encounter.year": 2026.0}}]}))
+ server = Server(state)
+ argv = ["client", "--base-url", server.base, "--state", str(state), "--rows", str(rows), "--media-dir", root]
+ with patch.object(sys, "argv", argv), patch.dict(os.environ, {"WILDBOOK_SUBMISSIONS_TOKEN": "test"}), patch.object(client, "Client", return_value=server), patch.object(client.time, "sleep"), patch("builtins.print"):
+ self.assertEqual(0, client.main()) # validate a whole-number float
+ argv.append("--commit")
+ self.assertEqual(0, client.main()) # resume after server normalized it to int
+ self.assertEqual(0, client.main()) # accepted execution is not repeated
+ self.assertEqual(2, len(server.creates))
+ self.assertEqual(server.creates[0], server.creates[1])
+ self.assertEqual(1, len(server.commits))
+ self.assertEqual(4, len(server.pages))
+ self.assertEqual("original-operation", json.loads(state.read_text())["operationId"])
+ self.assertNotIn("test", state.read_text())
+
+ def test_main_corrects_rows_in_same_draft_and_preserves_actionable_error(self):
+ import json
+ import os
+ import sys
+ class Server:
+ base = "http://localhost"
+ def __init__(self, old):
+ self.rows = old["rows"]
+ self.reject = False
+ self.puts = 0
+ def request(self, method, path, data=None, headers=None):
+ if path.endswith("/capabilities"):
+ return {"admissionEnabled": True, "commitEnabled": True, "stagingAvailable": True, "limits": {"maxFileBytes": 10000}}
+ if path.endswith("/rows"):
+ if method == "PUT":
+ self.puts += 1
+ if self.reject:
+ raise client.ApiError(422, "precise field error")
+ self.rows = data["rows"]
+ return {"rows": self.rows, "revision": 1}
+ if path.endswith("/validate"):
+ return {"valid": True, "errors": [], "revision": 1, "id": "v"}
+ return {"state": "draft", "revision": 1}
+ with tempfile.TemporaryDirectory() as root:
+ state, rows = Path(root) / "state.json", Path(root) / "rows.json"
+ old = {"rows": [{"clientRowId": "one", "fields": {"Encounter.year": 2025}}]}
+ new = {"rows": [{"clientRowId": "one", "fields": {"Encounter.year": 2026}}]}
+ client.save(state, {"id": "same-draft", "rowsDigest": client.digest(old), "baseUrl": "http://localhost", "source": "submissions-reference-client"})
+ rows.write_text(json.dumps(new))
+ server = Server(old)
+ argv = ["client", "--base-url", server.base, "--state", str(state), "--rows", str(rows), "--media-dir", root]
+ with patch.object(sys, "argv", argv), patch.dict(os.environ, {"WILDBOOK_SUBMISSIONS_TOKEN": "test"}), patch.object(client, "Client", return_value=server), patch("builtins.print"):
+ server.reject = True
+ with self.assertRaisesRegex(client.ApiError, "precise field error"):
+ client.main()
+ server.reject = False
+ self.assertEqual(0, client.main())
+ self.assertEqual("same-draft", json.loads(state.read_text())["id"])
+ self.assertEqual(client.digest(new), json.loads(state.read_text())["rowsDigest"])
+
+ def test_main_refuses_concurrent_state_use(self):
+ import os
+ import sys
+ import fcntl
+ with tempfile.TemporaryDirectory() as root:
+ state = Path(root) / "state.json"
+ fd = os.open(str(state) + ".lock", os.O_CREAT | os.O_WRONLY, 0o600)
+ try:
+ fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
+ with patch.object(sys, "argv", ["client", "--base-url", "http://localhost", "--state", str(state), "--cancel"]), patch.dict(os.environ, {"WILDBOOK_SUBMISSIONS_TOKEN": "test"}):
+ with self.assertRaisesRegex(ValueError, "Another client"):
+ client.main()
+ finally:
+ os.close(fd)
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/src/main/java/org/ecocean/StartupWildbook.java b/src/main/java/org/ecocean/StartupWildbook.java
index 909fce4e68..5b718f4530 100644
--- a/src/main/java/org/ecocean/StartupWildbook.java
+++ b/src/main/java/org/ecocean/StartupWildbook.java
@@ -156,6 +156,8 @@ public static void ensureProfilePhotoKeywordExists(Shepherd myShepherd) {
}
// these get run with each tomcat startup/shutdown, if web.xml is configured accordingly. see, e.g. https://stackoverflow.com/a/785802
+ private org.ecocean.api.submission.SubmissionWorker submissionWorker;
+
public void contextInitialized(ServletContextEvent sce) {
ServletContext sContext = sce.getServletContext();
String context = "context0";
@@ -222,6 +224,8 @@ public void contextInitialized(ServletContextEvent sce) {
} catch (Exception f) {
f.printStackTrace();
} finally { myShepherd.rollbackAndClose(); }
+ if (org.ecocean.api.submission.SubmissionPolicy.workerEnabled(context))
+ submissionWorker = new org.ecocean.api.submission.SubmissionWorker(sContext);
}
private void startIAQueues(String context) {
@@ -901,6 +905,7 @@ public void contextDestroyed(ServletContextEvent sce) {
// nulling the executor handle and waits up to 15s for in-flight
// ticks; any tick still running after that gets shutdownNow().
// The poll loop's interrupt/null checks make subsequent work bail.
+ if (submissionWorker != null) submissionWorker.close();
shutdownWbiaRegisterExecutor();
AnnotationLite.cleanup(sContext, context);
QueueUtil.cleanup();
diff --git a/src/main/java/org/ecocean/api/AgentSkill.java b/src/main/java/org/ecocean/api/AgentSkill.java
index 866d65b8a2..908b691aa3 100644
--- a/src/main/java/org/ecocean/api/AgentSkill.java
+++ b/src/main/java/org/ecocean/api/AgentSkill.java
@@ -33,6 +33,7 @@ public class AgentSkill extends ApiBase {
m.put("how-good-is-our-matching", "how-good-is-our-matching.md");
m.put("review-id-problems", "review-id-problems.md");
m.put("inat-to-wildbook-import", "inat-to-wildbook-import.md");
+ m.put("submit-sightings", "submit-sightings.md");
SKILL_RESOURCES = Collections.unmodifiableMap(m);
}
diff --git a/src/main/java/org/ecocean/api/AuthToken.java b/src/main/java/org/ecocean/api/AuthToken.java
index 024c9d063c..4121b36f7e 100644
--- a/src/main/java/org/ecocean/api/AuthToken.java
+++ b/src/main/java/org/ecocean/api/AuthToken.java
@@ -64,12 +64,35 @@ private void handle(HttpServletRequest request, HttpServletResponse response) th
return;
}
long ttl = ttlFromConfig(tokenContext);
- String token = jwt.sign(user.getId(), tokenContext, ttl);
+ String scope = request.getParameter("scope");
+ if (scope != null) {
+ if (!org.ecocean.api.submission.SubmissionPolicy.READ.equals(scope)
+ && !org.ecocean.api.submission.SubmissionPolicy.WRITE.equals(scope)) {
+ writeError(response, 400, "unsupported scope");
+ return;
+ }
+ // Basic credentials are resolved in context0; do not grant a capability for another context.
+ if (!context.equals(tokenContext)) {
+ writeError(response, 503, "submission token context unavailable");
+ return;
+ }
+ if (org.ecocean.api.submission.SubmissionPolicy.WRITE.equals(scope)) {
+ try {
+ org.ecocean.api.submission.SubmissionPolicy.requireAdmission(context, user.getId());
+ } catch (org.ecocean.api.submission.SubmissionException ex) {
+ writeError(response, ex.status, ex.getMessage());
+ return;
+ }
+ }
+ }
+ String token = scope == null ? jwt.sign(user.getId(), tokenContext, ttl)
+ : jwt.signSubmission(user.getId(), tokenContext, ttl, scope);
System.out.println("AuthToken mint OK user=" + username + " ip=" + clientIp);
JSONObject out = new JSONObject();
out.put("token", token);
out.put("tokenType", "Bearer");
out.put("expiresInSeconds", ttl / 1000L);
+ if (scope != null) out.put("scope", scope);
response.setStatus(200);
response.setContentType("application/json");
response.getWriter().write(out.toString());
diff --git a/src/main/java/org/ecocean/api/Submissions.java b/src/main/java/org/ecocean/api/Submissions.java
new file mode 100644
index 0000000000..442e3c514a
--- /dev/null
+++ b/src/main/java/org/ecocean/api/Submissions.java
@@ -0,0 +1,122 @@
+package org.ecocean.api;
+
+import java.io.IOException;
+import java.nio.charset.StandardCharsets;
+import javax.servlet.http.HttpServletRequest;
+import javax.servlet.http.HttpServletResponse;
+import org.ecocean.api.submission.*;
+import org.ecocean.security.SubmissionAuthenticationFilter;
+import org.ecocean.security.SubmissionAuthenticationFilter.Actor;
+import org.json.JSONObject;
+
+/** HTTP adapter for the gated, Bearer-only submission draft pilot. */
+public class Submissions extends ApiBase {
+ @Override protected void service(HttpServletRequest request, HttpServletResponse response) throws IOException {
+ response.setHeader("Cache-Control", "no-store");
+ response.setContentType("application/json;charset=UTF-8");
+ try {
+ Actor actor = (Actor)request.getAttribute(SubmissionAuthenticationFilter.ACTOR);
+ if (actor == null) throw new SubmissionException(401, "AUTHENTICATION_REQUIRED", "Authentication required");
+ String path = request.getPathInfo();
+ if (path == null || path.equals("/")) path = "";
+ String method = request.getMethod();
+ if (!method.equals("GET")) SubmissionPolicy.requireAdmission("context0", actor.id);
+ SubmissionStore store = store();
+ JSONObject result = null;
+ if (path.isEmpty() && method.equals("POST")) {
+ result = store.create("context0", actor.id, request.getHeader("Idempotency-Key"), body(request));
+ response.setStatus(201);
+ response.setHeader("Location", request.getContextPath() + "/api/v3/submissions/" + result.getString("id"));
+ } else if (path.equals("/capabilities") && method.equals("GET")) {
+ result = capabilities();
+ } else {
+ String[] parts = path.split("/", -1);
+ if (parts.length < 2 || !org.ecocean.Util.isUUID(parts[1])) throw new SubmissionException(404, "NOT_FOUND", "Route not found");
+ String id = parts[1];
+ if (parts.length == 2 && method.equals("GET")) result = store.get("context0", actor.id, id, actor.admin, false);
+ else if (parts.length == 2 && method.equals("DELETE")) {
+ store.cancel("context0", actor.id, id, actor.admin, revision(request));
+ response.setStatus(204); return;
+ } else if (parts.length == 3 && parts[2].equals("rows") && method.equals("GET"))
+ result = store.get("context0", actor.id, id, actor.admin, true);
+ else if (parts.length == 3 && parts[2].equals("rows") && method.equals("PUT"))
+ result = store.replaceRows("context0", actor.id, id, actor.admin, revision(request), body(request));
+ else if (parts.length == 3 && parts[2].equals("files") && method.equals("GET"))
+ result = store.manifest("context0", actor.id, id, actor.admin);
+ else if (parts.length == 3 && parts[2].equals("files") && method.equals("POST")) {
+ long rev = revision(request);
+ SubmissionFiles storage = new SubmissionFiles("context0", getServletContext());
+ result = store.upload("context0", actor.id, id, actor.admin, rev, storage, limit -> storage.receive(request, limit));
+ } else if (parts.length == 3 && parts[2].equals("validate") && method.equals("POST")) {
+ long rev = revision(request); SubmissionJson.keys(body(request));
+ result = store.validate("context0", actor.id, id, actor.admin, rev, new SubmissionFiles("context0", getServletContext()));
+ } else if (parts.length == 3 && parts[2].equals("commit") && method.equals("POST")) {
+ result = new SubmissionJobs("context0").enqueue("context0", actor.id, id, actor.admin,
+ revision(request), request.getHeader("Idempotency-Key"), body(request));
+ response.setStatus(202); response.setHeader("Location", request.getContextPath() + "/api/v3/submissions/" + id);
+ } else if (parts.length == 3 && parts[2].equals("results") && method.equals("GET")) {
+ int offset = page(request.getParameter("cursor"), 0), limit = page(request.getParameter("limit"), 100);
+ result = new SubmissionJobs("context0").results("context0", actor.id, id, actor.admin, offset, limit);
+ } else throw new SubmissionException(404, "NOT_FOUND", "Route not available");
+ }
+ String contextPath = request.getContextPath() == null ? "" : request.getContextPath();
+ if (result.has("statusUrl")) result.put("statusUrl", contextPath + result.getString("statusUrl"));
+ if (result.has("links") && result.getJSONObject("links").has("importTask")) {
+ JSONObject links = result.getJSONObject("links"); links.put("importTask", contextPath + links.getString("importTask"));
+ }
+ if (result.has("revision")) response.setHeader("ETag", "\"" + result.getLong("revision") + "\"");
+ response.getWriter().write(result.toString());
+ } catch (SubmissionException ex) { SubmissionAuthenticationFilter.error(response, ex); }
+ catch (org.apache.commons.fileupload.FileUploadBase.SizeLimitExceededException | org.apache.commons.fileupload.FileUploadBase.FileSizeLimitExceededException ex) { SubmissionAuthenticationFilter.error(response, new SubmissionException(413, "LIMIT_EXCEEDED", "Multipart size limit exceeded")); }
+ catch (org.json.JSONException ex) { SubmissionAuthenticationFilter.error(response, new SubmissionException(400, "BAD_REQUEST", "Invalid JSON")); }
+ catch (Exception ex) {
+ getServletContext().log("Submissions request failed", ex);
+ SubmissionAuthenticationFilter.error(response, new SubmissionException(500, "INTERNAL_ERROR", "Submission operation failed"));
+ }
+ }
+ private int page(String value, int fallback) {
+ if (value == null) return fallback;
+ if (!value.matches("[0-9]{1,8}")) throw new SubmissionException(400, "BAD_REQUEST", "Invalid pagination value");
+ return Integer.parseInt(value);
+ }
+ protected SubmissionStore store() { return new SubmissionStore("context0"); }
+ private JSONObject body(HttpServletRequest request) throws IOException {
+ if (request.getContentType() == null || !request.getContentType().split(";")[0].trim().equalsIgnoreCase("application/json"))
+ throw new SubmissionException(400, "BAD_REQUEST", "application/json required");
+ byte[] bytes = request.getInputStream().readNBytes(SubmissionPolicy.MAX_BODY_BYTES + 1);
+ if (bytes.length > SubmissionPolicy.MAX_BODY_BYTES) throw new SubmissionException(413, "LIMIT_EXCEEDED", "Request body limit exceeded");
+ return SubmissionJson.parse(bytes);
+ }
+ private long revision(HttpServletRequest request) {
+ String value = request.getHeader("If-Match");
+ if (value == null) throw new SubmissionException(428, "PRECONDITION_REQUIRED", "If-Match required");
+ if (!value.matches("\"[0-9]{1,18}\"")) throw new SubmissionException(400, "BAD_REQUEST", "Invalid If-Match revision");
+ return Long.parseLong(value.substring(1, value.length() - 1));
+ }
+ private JSONObject capabilities() {
+ boolean staging = false;
+ try { new SubmissionFiles("context0", getServletContext()); staging = true; }
+ catch (RuntimeException unavailable) { /* discovery remains available before storage configuration */ }
+ return new JSONObject().put("contractVersion", "1")
+ .put("admissionEnabled", SubmissionPolicy.enabled("context0"))
+ .put("stagingAvailable", staging)
+ .put("commitEnabled", SubmissionPolicy.commitEnabled("context0")).put("authentication", new org.json.JSONArray().put("bearer"))
+ .put("processingModes", new org.json.JSONArray().put("import-only"))
+ .put("operations", new org.json.JSONArray().put("create").put("get").put("replace-rows").put("get-rows").put("cancel").put("upload").put("get-files").put("validate").put("commit").put("results"))
+ .put("limits", new JSONObject().put("maxRows", SubmissionPolicy.MAX_ROWS)
+ .put("maxFieldsPerRow", 256)
+ .put("maxRequestBytes", SubmissionPolicy.MAX_BODY_BYTES).put("maxDraftsPerUser", 20).put("maxNewDraftsPerDay", 20)
+ .put("maxFileBytes", SubmissionFiles.maxFileBytes("context0"))
+ .put("maxDraftBytes", SubmissionFiles.MAX_DRAFT_BYTES)
+ .put("maxMediaPerEncounter", Math.max(1, org.ecocean.CommonConfiguration.getMaxMediaCountEncounter("context0")))
+ .put("maxActiveJobs", 1)
+ .put("draftTtlSeconds", SubmissionPolicy.DRAFT_TTL_MILLIS / 1000)
+ .put("idempotencyRetentionSeconds", SubmissionPolicy.DRAFT_TTL_MILLIS / 1000))
+ .put("rowFields", new JSONObject().put("supported", new org.json.JSONArray(new java.util.TreeSet<>(SubmissionValidator.FIELDS)))
+ .put("indexedMedia", "Encounter.mediaAsset0 through Encounter.mediaAsset199")
+ .put("required", new org.json.JSONArray().put("Encounter.genus").put("Encounter.specificEpithet")
+ .put("Encounter.year").put("Encounter.locationID")))
+ .put("uploadMediaTypes", new org.json.JSONArray().put("image/jpeg").put("image/png"))
+ .put("maxImagePixels", SubmissionFiles.MAX_PIXELS).put("maxFiles", SubmissionFiles.MAX_FILES);
+ }
+}
diff --git a/src/main/java/org/ecocean/api/auth/JwtService.java b/src/main/java/org/ecocean/api/auth/JwtService.java
index 5f093f3376..b2159078d1 100644
--- a/src/main/java/org/ecocean/api/auth/JwtService.java
+++ b/src/main/java/org/ecocean/api/auth/JwtService.java
@@ -17,9 +17,10 @@
import org.ecocean.Util;
/**
- * Issues and verifies short-lived RS256 JWTs that carry ONLY identity
- * (subject = user UUID, context). No admin/role claims — the consumer
- * resolves privileges fresh. Wildbook holds the private (signing) key;
+ * Issues and verifies short-lived RS256 JWTs carrying identity
+ * (subject = user UUID, context), and an optional explicitly issued submission
+ * capability. No admin/role claims — the consumer resolves privileges fresh.
+ * Wildbook holds the private (signing) key;
* the external scoped-access kernel holds the public key.
*
* Keys are RSA, supplied as Base64 of the encoded key bytes (private = PKCS8,
@@ -97,25 +98,46 @@ public boolean canVerify() {
}
public String sign(String userUuid, String context, long ttlMillis) {
+ return sign(userUuid, context, ttlMillis, null);
+ }
+
+ /** Explicitly requested submission capability; existing issuance stays identity-only. */
+ public String signSubmission(String userUuid, String context, long ttlMillis, String scope) {
+ if (!org.ecocean.api.submission.SubmissionPolicy.READ.equals(scope)
+ && !org.ecocean.api.submission.SubmissionPolicy.WRITE.equals(scope))
+ throw new IllegalArgumentException("Invalid submission scope");
+ return sign(userUuid, context, ttlMillis, scope);
+ }
+
+ private String sign(String userUuid, String context, long ttlMillis, String scope) {
if (!isEnabled()) throw new IllegalStateException("JwtService not enabled (no private key)");
long now = System.currentTimeMillis();
io.jsonwebtoken.JwtBuilder b = Jwts.builder()
.issuer(issuer)
- .audience().add(audience).and()
+ .audience().add(scope == null ? audience : audience + "/submissions").and()
.subject(userUuid)
.claim("context", context)
.id(Util.generateUUID())
.issuedAt(new Date(now))
.expiration(new Date(now + ttlMillis));
+ if (scope != null) b.claim("submissionScope", scope);
if (Util.stringExists(keyId)) b.header().keyId(keyId).and(); // 'kid' for rotation
return b.signWith(privateKey, Jwts.SIG.RS256).compact();
}
public Jws verify(String token) {
+ return verify(token, audience);
+ }
+
+ public Jws verifySubmission(String token) {
+ return verify(token, audience + "/submissions");
+ }
+
+ private Jws verify(String token, String expectedAudience) {
if (publicKey == null) throw new IllegalStateException("JwtService cannot verify (no public key)");
Jws jws = Jwts.parser()
.requireIssuer(issuer)
- .requireAudience(audience)
+ .requireAudience(expectedAudience)
.verifyWith(publicKey)
.build()
.parseSignedClaims(token);
diff --git a/src/main/java/org/ecocean/api/bulk/BulkImporter.java b/src/main/java/org/ecocean/api/bulk/BulkImporter.java
index 871d180285..e4994a52df 100644
--- a/src/main/java/org/ecocean/api/bulk/BulkImporter.java
+++ b/src/main/java/org/ecocean/api/bulk/BulkImporter.java
@@ -42,6 +42,15 @@ public class BulkImporter {
private String importTaskId = null;
private Shepherd myShepherd = null;
private long startTime = -1l;
+ private boolean deferSideEffects = false;
+ private java.util.function.BiConsumer rowCollector;
+
+ /** Opt-in transaction boundary for submissions; existing callers retain legacy behavior. */
+ public BulkImporter deferSideEffects(java.util.function.BiConsumer collector) {
+ this.deferSideEffects = true;
+ this.rowCollector = collector;
+ return this;
+ }
// caching loaded and (more imporantly?) newly created objects, so they can be
// used across all rows. StandardImport seemed to do some caching *based on user*
@@ -98,12 +107,13 @@ public JSONObject createImport()
}
// } else if (fieldObj instanceof BulkValidatorException) {
}
- System.out.println("createImport() row=" + rowNum);
+ trace("createImport() row=" + rowNum);
try {
- processRow(fields);
+ Encounter resolved = processRow(fields);
+ if (rowCollector != null) rowCollector.accept(rowNum, resolved);
} catch (Exception ex) {
// TODO we could allow this some leeway with a tolerance setting
- System.out.println("createImport() row=" + rowNum + " failed with " + ex);
+ trace("createImport() row=" + rowNum + " failed with " + ex);
ex.printStackTrace();
throw new ServletException("unexpected exception on processRow for row=" + rowNum +
": " + ex);
@@ -113,7 +123,7 @@ public JSONObject createImport()
markProgress(rowNum, dataRows.size(), 0.2d, 0.5d);
}
logProgress("end processRows");
- System.out.println(
+ trace(
"------------ all rows processed; beginning persistence -------------\n");
int persistenceTicksTotal = mediaAssetMap.values().size() + userCache.values().size() +
encounterCache.values().size() + occurrenceCache.values().size() +
@@ -124,7 +134,7 @@ public JSONObject createImport()
for (MediaAsset ma : mediaAssetMap.values()) {
ma.setSkipAutoIndexing(true);
MediaAssetFactory.save(ma, myShepherd);
- System.out.println("MMMM " + ma);
+ trace("MMMM " + ma);
arr.put(ma.getIdInt());
maIds.add(ma.getIdInt());
// see note on MediaAsset.getSkipAutoIndexing()
@@ -142,8 +152,9 @@ public JSONObject createImport()
arr = new JSONArray();
for (Encounter enc : encounterCache.values()) {
// it is a certain kind of painful that if you do not pass id here it assigns a new random one
- myShepherd.storeNewEncounter(enc, enc.getId());
- System.out.println("EEEE " + enc);
+ if (deferSideEffects) { enc.setEncounterNumber(enc.getId()); myShepherd.getPM().makePersistent(enc); }
+ else myShepherd.storeNewEncounter(enc, enc.getId());
+ trace("EEEE " + enc);
arr.put(enc.getId());
needIndexing.add(enc);
persistenceTicks++;
@@ -153,8 +164,9 @@ public JSONObject createImport()
rtn.put("encounters", arr);
arr = new JSONArray();
for (Occurrence occ : occurrenceCache.values()) {
- myShepherd.storeNewOccurrence(occ);
- System.out.println("OOOO " + occ);
+ if (deferSideEffects) myShepherd.getPM().makePersistent(occ);
+ else myShepherd.storeNewOccurrence(occ);
+ trace("OOOO " + occ);
arr.put(occ.getId());
needIndexing.add(occ);
persistenceTicks++;
@@ -164,9 +176,10 @@ public JSONObject createImport()
rtn.put("sightings", arr);
arr = new JSONArray();
for (MarkedIndividual indiv : individualCache.values()) {
- myShepherd.storeNewMarkedIndividual(indiv);
+ if (deferSideEffects) myShepherd.getPM().makePersistent(indiv);
+ else myShepherd.storeNewMarkedIndividual(indiv);
indiv.refreshNamesCache();
- System.out.println("IIII " + indiv);
+ trace("IIII " + indiv);
arr.put(indiv.getId());
needIndexing.add(indiv);
persistenceTicks++;
@@ -175,17 +188,18 @@ public JSONObject createImport()
logProgress("end persist MarkedIndividual");
rtn.put("individuals", arr);
for (Project proj : projectCache.values()) {
- myShepherd.storeNewProject(proj);
- System.out.println("PPPP " + proj);
+ if (deferSideEffects) myShepherd.getPM().makePersistent(proj);
+ else myShepherd.storeNewProject(proj);
+ trace("PPPP " + proj);
persistenceTicks++;
markProgress(persistenceTicks, persistenceTicksTotal, 0.7d, 0.3d);
}
logProgress("persist COMPLETE");
- System.out.println(
+ trace(
"------------ persistence complete; background indexing and MA children -------------\n");
// clears shepherd/pmf cache, which we seem to do when we create encounters (?)
- myShepherd.cacheEvictAll();
- MediaAsset.updateStandardChildrenBackground(myShepherd.getContext(), maIds, new Runnable() {
+ if (!deferSideEffects) myShepherd.cacheEvictAll();
+ if (!deferSideEffects) MediaAsset.updateStandardChildrenBackground(myShepherd.getContext(), maIds, new Runnable() {
public void run() {
BulkImportUtil.bulkOpensearchIndex(needIndexing);
}
@@ -195,13 +209,13 @@ public void run() {
}
// this assumes all values have been validated, so just go for it! set data with values. good luck!
- private void processRow(List fields) {
+ private Encounter processRow(List fields) {
// some fields we do on a subsequent pass, as they require special care
// handy for these subsequent passes
Map fmap = new HashMap();
for (BulkValidator field : fields) {
- System.out.println(" >> " + field);
+ trace(" >> " + field);
fmap.put(field.getFieldName(), field);
}
Set allFieldNames = fmap.keySet();
@@ -368,7 +382,7 @@ private void processRow(List fields) {
String munit = null;
if (i < munits.size()) munit = munits.get(i);
Measurement meas = new Measurement(enc.getId(), mvals.get(i), mdbl, munit, sampProt);
- System.out.println("[INFO] field " + measFN.get(i) + " [i=" + i + "] created " + meas);
+ trace("[INFO] field " + measFN.get(i) + " [i=" + i + "] created " + meas);
enc.setMeasurement(meas);
}
handleSocialUnit(indiv, fmap.get("SocialUnit.socialUnitName"), fmap.get("Membership.role"));
@@ -380,7 +394,7 @@ private void processRow(List fields) {
setting the value on the Encouner only. so we follow this as represented in that class, fbow.
*/
for (BulkValidator bv : fields) {
- System.out.println("bv>>>> " + bv);
+ trace("bv>>>> " + bv);
String fieldName = bv.getFieldName();
switch (fieldName) {
case "Encounter.latitude":
@@ -678,7 +692,7 @@ private void processRow(List fields) {
case "Sighting.taxonomy0":
case "Taxonomy.commonName":
case "Taxonomy.scientificName":
- System.out.println("[INFO] " + fieldName + " currently not implemented");
+ trace("[INFO] " + fieldName + " currently not implemented");
break;
*/
@@ -689,12 +703,12 @@ private void processRow(List fields) {
//case "Sighting.numSubFemales":
*/
default:
- System.out.println("[INFO] processRow() ignored a field [" + fieldName +
+ trace("[INFO] processRow() ignored a field [" + fieldName +
"] that was flagged valid");
}
}
// fields done
- System.out.println("+ populated data on " + enc);
+ trace("+ populated data on " + enc);
// now attach annotations
String tx = enc.getTaxonomyString();
List annots = new ArrayList();
@@ -715,7 +729,7 @@ private void processRow(List fields) {
// image, so we advance `offset` to consume its keyword/quality
// slot — otherwise a later valid image would inherit this
// corrupt image's positional metadata.
- System.out.println("[WARN] processRow: skipping image with no MediaAsset (likely "
+ trace("[WARN] processRow: skipping image with no MediaAsset (likely "
+ "corrupt/unreadable) for maKey=" + maKey + ", value=" + bv.getValueString());
offset++;
continue;
@@ -738,11 +752,12 @@ private void processRow(List fields) {
offset++;
}
if (annots.size() > 0) enc.addAnnotations(annots);
- System.out.println("+ populated " + annots.size() + " MediaAssets on " + enc);
+ trace("+ populated " + annots.size() + " MediaAssets on " + enc);
+ return enc;
}
public void markProgress(int ticks, int total, double base, double weight) {
- if (this.importTaskId == null) return;
+ if (this.importTaskId == null || deferSideEffects) return;
// we want our own shepherd here so we can persist this task independent of our main shepherd
Shepherd taskShepherd = new Shepherd(this.myShepherd.getContext());
taskShepherd.setAction("BulkImporter.markProgress");
@@ -845,11 +860,11 @@ private void handleSamples(Encounter enc, Map fmap) {
ex.printStackTrace();
}
if ((all0 == null) || (all1 == null)) {
- System.out.println(
+ trace(
"BulkImporter.handleSamples(): failed to get allele ints for " +
zeros[i] + "; " + ones[i]);
} else if (names[i].equals("")) {
- System.out.println("BulkImporter.handleSamples(): empty name for i=" + i +
+ trace("BulkImporter.handleSamples(): empty name for i=" + i +
" in " + alleleNames);
} else {
Locus locus = new Locus(names[i], all0, all1);
@@ -857,7 +872,7 @@ private void handleSamples(Encounter enc, Map fmap) {
}
}
} else {
- System.out.println("BulkImporter.handleSamples(): length mismatch for (" +
+ trace("BulkImporter.handleSamples(): length mismatch for (" +
alleleNames + "|" + alleleZeros + "|" + alleleOnes + ")");
}
if (loci.size() > 0) {
@@ -865,7 +880,7 @@ private void handleSamples(Encounter enc, Map fmap) {
Util.generateUUID(), tsId, enc.getId(), loci);
myShepherd.getPM().makePersistent(markers);
sample.addGeneticAnalysis(markers);
- System.out.println("BulkImporter.handleSamples(): adding " + markers + " to " +
+ trace("BulkImporter.handleSamples(): adding " + markers + " to " +
sample);
}
}
@@ -877,7 +892,7 @@ private void handleSamples(Encounter enc, Map fmap) {
SexAnalysis sexAn = new SexAnalysis(Util.generateUUID(), sas, enc.getId(), tsId);
myShepherd.getPM().makePersistent(sexAn);
sample.addGeneticAnalysis(sexAn);
- System.out.println("BulkImporter.handleSamples(): adding " + sexAn + " to " + sample);
+ trace("BulkImporter.handleSamples(): adding " + sexAn + " to " + sample);
}
// haplotype
String hap = null;
@@ -888,7 +903,7 @@ private void handleSamples(Encounter enc, Map fmap) {
enc.getId(), tsId);
myShepherd.getPM().makePersistent(mda);
sample.addGeneticAnalysis(mda);
- System.out.println("BulkImporter.handleSamples(): adding " + mda + " to " + sample);
+ trace("BulkImporter.handleSamples(): adding " + mda + " to " + sample);
}
// wrap it up, we are done!
enc.addTissueSample(sample);
@@ -906,7 +921,7 @@ private MarkedIndividual getOrCreateMarkedIndividual(String id,
MarkedIndividual indiv = myShepherd.getMarkedIndividual(id);
if (!(fmap.containsKey("Encounter.genus") &&
fmap.containsKey("Encounter.specificEpithet"))) {
- System.out.println("[WARNING] BulkImporter.getOrCreateMarkedIndividual(" + id +
+ trace("[WARNING] BulkImporter.getOrCreateMarkedIndividual(" + id +
") is missing genus and/or specificEpithet values");
return null;
}
@@ -925,7 +940,7 @@ private MarkedIndividual getOrCreateMarkedIndividual(String id,
indiv.setSpecificEpithet(specificEpithet);
indiv.setVersion();
// TODO what else???
- System.out.println(
+ trace(
"[INFO] BulkImporter.getOrCreateMarkedIndividual() creating new; could not find existing indiv based on id="
+ id + " => " + indiv);
}
@@ -979,7 +994,7 @@ private User getOrCreateUser(String email, String fullname, String affiliation)
user = new User(email, Util.generateUUID());
user.setFullName(fullname);
user.setAffiliation(affiliation);
- System.out.println("[INFO] BulkImporter.getOrCreateUser() creating new " + user);
+ trace("[INFO] BulkImporter.getOrCreateUser() creating new " + user);
}
userCache.put(email, user);
return user;
@@ -999,7 +1014,7 @@ private Project getOrCreateProject(String projectPrefix, String projectName,
if (proj == null) {
proj = myShepherd.getProjectByProjectIdPrefixPrefix(projectPrefix);
if (proj != null)
- System.out.println(
+ trace(
"[INFO] BulkImporter.getOrCreateProject() fuzzy-matched projectPrefix '" +
projectPrefix + "' to " + proj);
}
@@ -1045,6 +1060,8 @@ private Occurrence getOrCreateOccurrence(Map fmap) {
return occ;
}
+ private void trace(String text) { if (!deferSideEffects) System.out.println(text); }
+
public static void logProgress(String id, String msg, Long startTime) {
Util.mark("BulkImporter.logProgress[" + id + "]: " + msg, startTime);
}
diff --git a/src/main/java/org/ecocean/api/submission/SubmissionException.java b/src/main/java/org/ecocean/api/submission/SubmissionException.java
new file mode 100644
index 0000000000..5f6cbf239f
--- /dev/null
+++ b/src/main/java/org/ecocean/api/submission/SubmissionException.java
@@ -0,0 +1,11 @@
+package org.ecocean.api.submission;
+
+public class SubmissionException extends RuntimeException {
+ public final int status;
+ public final String code;
+ public SubmissionException(int status, String code, String message) {
+ super(message);
+ this.status = status;
+ this.code = code;
+ }
+}
diff --git a/src/main/java/org/ecocean/api/submission/SubmissionFiles.java b/src/main/java/org/ecocean/api/submission/SubmissionFiles.java
new file mode 100644
index 0000000000..a3abf6e288
--- /dev/null
+++ b/src/main/java/org/ecocean/api/submission/SubmissionFiles.java
@@ -0,0 +1,200 @@
+package org.ecocean.api.submission;
+
+import java.io.*;
+import java.nio.file.*;
+import java.security.*;
+import java.util.*;
+import javax.imageio.*;
+import javax.imageio.stream.ImageInputStream;
+import javax.servlet.http.HttpServletRequest;
+import org.apache.commons.fileupload.*;
+import org.apache.commons.fileupload.servlet.ServletFileUpload;
+import org.ecocean.CommonConfiguration;
+import org.ecocean.resumableupload.UploadPaths;
+import org.ecocean.servlet.ServletUtilities;
+import org.json.*;
+
+/** Private immutable staging. No client-controlled storage paths are accepted. */
+public class SubmissionFiles {
+ public static final long MAX_DRAFT_BYTES = 200L * 1024 * 1024;
+ public static final int MAX_FILES = 200;
+ public static final long MAX_PIXELS = 24_000_000;
+ private final Path root;
+ public SubmissionFiles(String context, javax.servlet.ServletContext servlet) {
+ this(configuredRoot(context, servlet));
+ }
+ public SubmissionFiles(Path root) {
+ this.root = root.toAbsolutePath().normalize();
+ }
+ private static Path configuredRoot(String context, javax.servlet.ServletContext servlet) {
+ String value = CommonConfiguration.getApiAccessProperty("submissions.stagingDirectory", context);
+ if (value == null || !Path.of(value).isAbsolute())
+ throw new SubmissionException(503, "CAPABILITY_UNAVAILABLE", "Private staging directory is not configured");
+ try {
+ Path root = Path.of(value).toRealPath();
+ Path legacy = new File(CommonConfiguration.getUploadTmpDir(context)).getCanonicalFile().toPath();
+ if (root.startsWith(legacy) || legacy.startsWith(root))
+ throw new SubmissionException(503, "CAPABILITY_UNAVAILABLE", "Private staging must be separate from legacy uploads");
+ String web = servlet.getRealPath("/");
+ if (web == null) throw new SubmissionException(503, "CAPABILITY_UNAVAILABLE", "Cannot verify private staging against web root");
+ rejectOverlap(root, Path.of(web).toFile().getCanonicalFile().toPath().getParent());
+ String imports = CommonConfiguration.getImportDir(context);
+ if (imports != null) rejectOverlap(root, new File(imports).getCanonicalFile().toPath());
+ org.ecocean.shepherd.core.Shepherd sh = new org.ecocean.shepherd.core.Shepherd(context);
+ try {
+ sh.beginDBTransaction();
+ java.util.List stores = org.ecocean.media.AssetStoreFactory.getStores(sh);
+ if (stores == null) throw new SubmissionException(503, "CAPABILITY_UNAVAILABLE", "Cannot verify asset-store boundaries");
+ for (org.ecocean.media.AssetStore store : stores) if (store instanceof org.ecocean.media.LocalAssetStore)
+ rejectOverlap(root, ((org.ecocean.media.LocalAssetStore)store).root().toFile().getCanonicalFile().toPath());
+ } finally { sh.rollbackAndClose(); }
+ return root;
+ } catch (IOException ex) { throw new SubmissionException(503, "CAPABILITY_UNAVAILABLE", "Private staging directory unavailable"); }
+ }
+ public static void rejectOverlap(Path root, Path served) {
+ if (served == null || root.startsWith(served) || served.startsWith(root))
+ throw new SubmissionException(503, "CAPABILITY_UNAVAILABLE", "Staging must be outside served and import directories");
+ }
+ public static long maxFileBytes(String context) {
+ return Math.min(MAX_DRAFT_BYTES, Math.max(1L, CommonConfiguration.getMaxMediaSizeInMegabytes(context)) * 1024 * 1024);
+ }
+ public static void checkName(String name) {
+ if (!UploadPaths.isSingleComponentName(name) || name.length() > 128 || !name.matches("[A-Za-z0-9][A-Za-z0-9_.-]*")
+ || !name.equals(ServletUtilities.cleanFileName(name)))
+ throw new SubmissionException(400, "BAD_REQUEST", "Use a filename of at most 128 ASCII letters, digits, dots, underscores or hyphens, starting with a letter or digit");
+ }
+ public JSONObject receive(HttpServletRequest request, long limit) throws Exception {
+ try { return receiveMultipart(request, limit); }
+ catch (FileUploadBase.FileUploadIOException ex) {
+ Throwable cause = ex.getCause();
+ if (cause instanceof FileUploadBase.SizeLimitExceededException || cause instanceof FileUploadBase.FileSizeLimitExceededException) throw new SubmissionException(413, "LIMIT_EXCEEDED", "Multipart size limit exceeded");
+ throw new SubmissionException(400, "BAD_REQUEST", "Malformed multipart upload");
+ } catch (FileUploadBase.SizeLimitExceededException | FileUploadBase.FileSizeLimitExceededException ex) { throw new SubmissionException(413, "LIMIT_EXCEEDED", "Multipart size limit exceeded"); }
+ catch (MultipartStream.MalformedStreamException ex) { throw new SubmissionException(400, "BAD_REQUEST", "Incomplete multipart body"); }
+ catch (FileUploadException | InvalidFileNameException ex) { throw new SubmissionException(400, "BAD_REQUEST", "Malformed multipart upload"); }
+ }
+ private JSONObject receiveMultipart(HttpServletRequest request, long limit) throws Exception {
+ if (!ServletFileUpload.isMultipartContent(request))
+ throw new SubmissionException(400, "BAD_REQUEST", "multipart/form-data with one file required");
+ ServletFileUpload upload = new ServletFileUpload();
+ upload.setSizeMax(limit + 65536); upload.setFileSizeMax(limit); upload.setHeaderEncoding("UTF-8");
+ FileItemIterator items = upload.getItemIterator(request);
+ if (!items.hasNext()) throw new SubmissionException(400, "BAD_REQUEST", "File required");
+ FileItemStream item = items.next();
+ if (item.isFormField() || !"file".equals(item.getFieldName()))
+ throw new SubmissionException(400, "BAD_REQUEST", "Exactly one file part named file required");
+ JSONObject entry = null;
+ try {
+ try (InputStream input = item.openStream()) { entry = write(item.getName(), input, limit); }
+ if (items.hasNext()) throw new SubmissionException(400, "BAD_REQUEST", "Exactly one file part required");
+ return entry;
+ } catch (Exception ex) { if (entry != null) try { remove(entry); } catch (IOException cleanup) { System.err.println("Submission temporary blob cleanup failed"); } throw ex; }
+ }
+ public JSONObject write(String name, InputStream input, long limit) throws Exception {
+ checkName(name);
+ if (Files.isSymbolicLink(root) || !Files.isDirectory(root, LinkOption.NOFOLLOW_LINKS))
+ throw new SubmissionException(503, "CAPABILITY_UNAVAILABLE", "Private staging unavailable");
+ boolean posix = Files.getFileStore(root).supportsFileAttributeView("posix");
+ Path candidate = root.resolve(UUID.randomUUID().toString());
+ Path dir = posix ? Files.createDirectory(candidate, java.nio.file.attribute.PosixFilePermissions.asFileAttribute(
+ java.nio.file.attribute.PosixFilePermissions.fromString("rwx------"))) : Files.createDirectory(candidate);
+ Path path = dir.resolve(name);
+ boolean complete = false;
+ try {
+ MessageDigest digest = MessageDigest.getInstance("SHA-256");
+ long count = 0; long deadline = System.nanoTime() + java.util.concurrent.TimeUnit.MINUTES.toNanos(2);
+ try (OutputStream out = posix ? java.nio.channels.Channels.newOutputStream(Files.newByteChannel(path,
+ java.util.EnumSet.of(StandardOpenOption.WRITE, StandardOpenOption.CREATE_NEW),
+ java.nio.file.attribute.PosixFilePermissions.asFileAttribute(java.nio.file.attribute.PosixFilePermissions.fromString("rw-------"))))
+ : Files.newOutputStream(path, StandardOpenOption.CREATE_NEW)) {
+ byte[] buffer = new byte[8192]; int n;
+ while ((n = input.read(buffer)) != -1) {
+ if (System.nanoTime() > deadline) throw new SubmissionException(408, "BAD_REQUEST", "Upload time limit exceeded");
+ count += n;
+ if (count > limit) throw new SubmissionException(413, "LIMIT_EXCEEDED", "File or remaining draft byte limit exceeded");
+ digest.update(buffer, 0, n); out.write(buffer, 0, n);
+ }
+ }
+ String mediaType = inspect(path);
+ String lower = name.toLowerCase(Locale.ROOT);
+ if (!(mediaType.equals("image/png") ? lower.endsWith(".png") : (lower.endsWith(".jpg") || lower.endsWith(".jpeg"))))
+ throw new SubmissionException(422, "VALIDATION_INVALID", "Filename extension must match image content");
+ StringBuilder sha = new StringBuilder(); for (byte b : digest.digest()) sha.append(String.format("%02x", b & 255));
+ JSONObject result = new JSONObject().put("name", name).put("sizeBytes", count).put("sha256", sha.toString())
+ .put("state", "complete").put("mediaType", mediaType).put("blob", dir.getFileName().toString());
+ complete = true; return result;
+ } finally { if (!complete) try { Files.deleteIfExists(path); Files.deleteIfExists(dir); } catch (IOException cleanup) { System.err.println("Submission temporary blob cleanup failed"); } }
+ }
+ public Path path(JSONObject entry) throws IOException {
+ String blob = entry.getString("blob"); String name = entry.getString("name"); checkName(name);
+ if (!org.ecocean.Util.isUUID(blob)) throw new IOException("Invalid blob identifier");
+ File dir = UploadPaths.resolveDirWithin(root.toFile(), blob);
+ if (dir == null || Files.isSymbolicLink(root.resolve(blob))) throw new IOException("Invalid staging path");
+ File file = UploadPaths.resolveWithin(dir, name);
+ if (file == null || Files.isSymbolicLink(dir.toPath().resolve(name))) throw new IOException("Invalid staging file");
+ return file.toPath();
+ }
+ public void verify(JSONObject entry) throws Exception {
+ Path path = path(entry);
+ MessageDigest digest = MessageDigest.getInstance("SHA-256"); long count = 0;
+ try (InputStream in = Files.newInputStream(path)) {
+ byte[] buf = new byte[8192]; int n;
+ while ((n = in.read(buf)) != -1) { count += n; if (count > MAX_DRAFT_BYTES) throw new IOException("Oversize staged file"); digest.update(buf, 0, n); }
+ }
+ StringBuilder sha = new StringBuilder(); for (byte b : digest.digest()) sha.append(String.format("%02x", b & 255));
+ if (count != entry.getLong("sizeBytes") || !sha.toString().equals(entry.getString("sha256"))) throw new IOException("Staged file changed");
+ inspect(path);
+ }
+ public static String inspect(Path path) throws IOException {
+ try (ImageInputStream input = ImageIO.createImageInputStream(path.toFile())) {
+ if (input == null) throw new IOException("Cannot read image");
+ Iterator readers = ImageIO.getImageReaders(input);
+ if (!readers.hasNext()) throw new SubmissionException(422, "VALIDATION_INVALID", "Unrecognized image");
+ ImageReader reader = readers.next();
+ try {
+ String format = reader.getFormatName().toLowerCase(Locale.ROOT);
+ if (!Set.of("jpeg", "jpg", "png").contains(format)) throw new SubmissionException(422, "VALIDATION_INVALID", "Only JPEG and PNG are supported");
+ reader.setInput(input, true, true);
+ int w = reader.getWidth(0), h = reader.getHeight(0);
+ if (w <= 0 || h <= 0 || w > 16000 || h > 16000 || (long)w * h > MAX_PIXELS)
+ throw new SubmissionException(413, "LIMIT_EXCEEDED", "Image exceeds dimension or pixel limit");
+ ImageReadParam param = reader.getDefaultReadParam();
+ param.setSourceSubsampling(4, 4, 0, 0);
+ java.awt.image.BufferedImage decoded = reader.read(0, param);
+ if (decoded == null) throw new IOException("Cannot decode image"); decoded.flush();
+ return format.equals("png") ? "image/png" : "image/jpeg";
+ } finally { reader.dispose(); }
+ } catch (IOException ex) { throw new SubmissionException(422, "VALIDATION_INVALID", "Image is truncated or cannot be decoded"); }
+ }
+ public void remove(JSONObject entry) throws IOException {
+ Path path = path(entry); Files.deleteIfExists(path); Files.deleteIfExists(path.getParent());
+ }
+ public static JSONArray publicFiles(JSONArray files) {
+ JSONArray result = new JSONArray();
+ for (int i = 0; i < files.length(); i++) { JSONObject file = new JSONObject(files.getJSONObject(i).toString()); file.remove("blob"); result.put(file); }
+ return result;
+ }
+ public void cleanup(Set retained) throws IOException {
+ long cutoff = System.currentTimeMillis() - SubmissionPolicy.DRAFT_TTL_MILLIS; int removed = 0;
+ try (DirectoryStream dirs = Files.newDirectoryStream(root)) {
+ for (Path dir : dirs) {
+ if (removed >= 5000) break;
+ try {
+ String blob = dir.getFileName().toString();
+ if (!org.ecocean.Util.isUUID(blob) || retained.contains(blob) || Files.isSymbolicLink(dir)
+ || !Files.isDirectory(dir, LinkOption.NOFOLLOW_LINKS) || Files.getLastModifiedTime(dir).toMillis() >= cutoff) continue;
+ java.util.List files = new ArrayList<>(); boolean safe = true;
+ try (DirectoryStream children = Files.newDirectoryStream(dir)) {
+ for (Path file : children) {
+ if (!Files.isRegularFile(file, LinkOption.NOFOLLOW_LINKS) || Files.getLastModifiedTime(file).toMillis() >= cutoff) { safe = false; break; }
+ files.add(file);
+ }
+ }
+ if (!safe) continue;
+ for (Path file : files) Files.deleteIfExists(file);
+ Files.deleteIfExists(dir); removed++;
+ } catch (IOException ex) { System.err.println("Submission blob cleanup deferred for one directory"); }
+ }
+ }
+ }
+}
diff --git a/src/main/java/org/ecocean/api/submission/SubmissionImporter.java b/src/main/java/org/ecocean/api/submission/SubmissionImporter.java
new file mode 100644
index 0000000000..4d9712b902
--- /dev/null
+++ b/src/main/java/org/ecocean/api/submission/SubmissionImporter.java
@@ -0,0 +1,67 @@
+package org.ecocean.api.submission;
+
+import java.util.*;
+import org.ecocean.*;
+import org.ecocean.api.UploadedFiles;
+import org.ecocean.api.bulk.*;
+import org.ecocean.media.MediaAsset;
+import org.ecocean.servlet.importer.ImportTask;
+import org.ecocean.shepherd.core.Shepherd;
+import org.ecocean.submission.Submission;
+import org.json.*;
+
+/** Caller owns the transaction. No commit, derivative or indexing dispatch occurs here. */
+public class SubmissionImporter {
+ /** Only thrown before any media copy or domain persistence begins. */
+ public static class PreImportRejection extends SubmissionException {
+ public PreImportRejection(int status, String code, String message) { super(status, code, message); }
+ }
+ public JSONObject execute(Submission draft, String taskId, Shepherd sh, SubmissionFiles storage) throws Exception {
+ User owner = sh.getUserByUUID(draft.getOwnerId());
+ if (owner == null || owner.getUsername() == null || owner.getUsername().isBlank() || !SubmissionPolicy.enrolled(draft.getContext(), draft.getOwnerId()))
+ throw new PreImportRejection(403, "ACCESS_DENIED", "Owner is no longer eligible");
+ JSONObject approved = new JSONObject(draft.getValidationJson());
+ JSONObject checked = new SubmissionValidator().validate(draft, sh, storage);
+ if (!approved.getBoolean("valid") || approved.getLong("revision") != draft.getRevision() || !checked.getBoolean("valid") || !approved.getString("configDigest").equals(checked.getString("configDigest"))
+ || !approved.getString("manifestDigest").equals(checked.getString("manifestDigest"))
+ || !SubmissionJson.canonical(approved.getJSONArray("normalizedRows")).equals(SubmissionJson.canonical(checked.getJSONArray("normalizedRows"))))
+ throw new PreImportRejection(409, "VALIDATION_STALE", "Input or configuration changed after validation");
+ ImportTask task = sh.getImportTask(taskId);
+ if (task == null) throw new PreImportRejection(409, "INVALID_STATE", "Reserved import task missing");
+ JSONArray rows = checked.getJSONArray("normalizedRows"), files = new JSONArray(draft.getFilesJson());
+ List
+
+ {t("API_TOKEN_PURPOSE")}
+ {
+ setPurpose(e.target.value);
+ setToken(null);
+ setExpiresIn(null);
+ setError(null);
+ }}>
+
+
+
+ {t(purpose === "import" ? "API_TOKEN_IMPORT_HELP" : "API_TOKEN_READ_HELP")}
+
+
{token && (
diff --git a/src/main/java/org/ecocean/Role.java b/src/main/java/org/ecocean/Role.java
index 215b5bd1b4..e7ccff2c87 100644
--- a/src/main/java/org/ecocean/Role.java
+++ b/src/main/java/org/ecocean/Role.java
@@ -27,9 +27,22 @@ public class Role implements java.io.Serializable {
public static final List SYSTEM_ROLES = Collections.unmodifiableList(
Arrays.asList("admin", "orgAdmin", "researcher", "rest", "machinelearning"));
- /** SYSTEM_ROLES as a set, for membership tests where the hierarchy order does not matter. */
+ /** Explicitly assigned capability; never included in bootstrap grants or merge ranking. */
+ public static final String API_SUBMISSION = "api-submission";
+
+ /** Reserved system names, including opt-in capabilities, excluded from location grants. */
public static final Set SYSTEM_ROLE_NAMES = Collections.unmodifiableSet(
- new LinkedHashSet(SYSTEM_ROLES));
+ reservedRoleNames());
+
+ private static Set reservedRoleNames() {
+ Set names = new LinkedHashSet(SYSTEM_ROLES);
+ names.add(API_SUBMISSION);
+ return names;
+ }
+
+ public static boolean canEditRole(String role, boolean siteAdmin) {
+ return siteAdmin || (!API_SUBMISSION.equals(role) && !"admin".equals(role));
+ }
private String username;
private String rolename;
diff --git a/src/main/java/org/ecocean/api/submission/SubmissionPolicy.java b/src/main/java/org/ecocean/api/submission/SubmissionPolicy.java
index 47d2709262..e75de41a87 100644
--- a/src/main/java/org/ecocean/api/submission/SubmissionPolicy.java
+++ b/src/main/java/org/ecocean/api/submission/SubmissionPolicy.java
@@ -1,6 +1,9 @@
package org.ecocean.api.submission;
-import java.util.Arrays;
+import javax.jdo.Query;
+import org.ecocean.Role;
+import org.ecocean.User;
+import org.ecocean.shepherd.core.Shepherd;
import org.ecocean.CommonConfiguration;
/** Installation-local pilot controls. Read access survives admission shutdown. */
@@ -20,12 +23,30 @@ public static boolean workerEnabled(String context) {
return "true".equalsIgnoreCase(CommonConfiguration.getApiAccessProperty("submissions.workerEnabled", context));
}
public static boolean enrolled(String context, String userId) {
- String users = CommonConfiguration.getApiAccessProperty("submissions.allowedUserIds", context);
- return userId != null && users != null && Arrays.stream(users.split(","))
- .map(String::trim).anyMatch(userId::equals);
+ if (userId == null) return false;
+ Shepherd sh = new Shepherd(context);
+ try {
+ sh.beginDBTransaction();
+ return enrolled(sh, userId);
+ } finally { sh.rollbackAndClose(); }
}
+
+ static boolean enrolled(Shepherd sh, String userId) {
+ if (userId == null) return false;
+ User user = sh.getUserByUUID(userId);
+ if (user == null || user.getUsername() == null || user.getUsername().isBlank()) return false;
+ // Query persisted grants on every admission check; JWT/session roles are not authority.
+ Query> query = sh.getPM().newQuery(Role.class,
+ "username == :username && rolename == :role && context == :context");
+ try {
+ query.setIgnoreCache(true);
+ query.setResult("count(this)");
+ return ((Number)query.execute(user.getUsername(), Role.API_SUBMISSION, sh.getContext())).longValue() > 0;
+ } finally { query.closeAll(); }
+ }
+
public static void requireAdmission(String context, String userId) {
if (!enabled(context)) throw new SubmissionException(503, "ADMISSION_DISABLED", "Submission admission is disabled");
- if (!enrolled(context, userId)) throw new SubmissionException(403, "ACCESS_DENIED", "Account is not enrolled in the pilot");
+ if (!enrolled(context, userId)) throw new SubmissionException(403, "ACCESS_DENIED", "Account requires the api-submission role");
}
}
diff --git a/src/main/java/org/ecocean/servlet/UserConsolidate.java b/src/main/java/org/ecocean/servlet/UserConsolidate.java
index 4cdbb902f7..e1dd4b2488 100644
--- a/src/main/java/org/ecocean/servlet/UserConsolidate.java
+++ b/src/main/java/org/ecocean/servlet/UserConsolidate.java
@@ -341,6 +341,11 @@ public static void consolidateRoles(Shepherd myShepherd, User userToRetain,
if (consolidatedUserRoles != null && consolidatedUserRoles.size() > 0) {
for (int i = 0; i < consolidatedUserRoles.size(); i++) {
Role currentRole = consolidatedUserRoles.get(i);
+ // Enrollment belongs to the retained account, not an automatically merged identity.
+ if (Role.API_SUBMISSION.equals(currentRole.getRolename())) {
+ myShepherd.getPM().deletePersistent(currentRole);
+ continue;
+ }
if (!retainedUserRoles.contains(currentRole)) {
// it's a new role for the retained user; add it. Note: this because the role usernames are different, this will in effect
// capture all retainedUserRoles. But since username is converted downstream, this is not actually a bug. Might could be
diff --git a/src/main/java/org/ecocean/servlet/UserCreate.java b/src/main/java/org/ecocean/servlet/UserCreate.java
index 5b71e4d038..625a003a4e 100644
--- a/src/main/java/org/ecocean/servlet/UserCreate.java
+++ b/src/main/java/org/ecocean/servlet/UserCreate.java
@@ -29,6 +29,31 @@ public void doGet(HttpServletRequest request, HttpServletResponse response)
doPost(request, response);
}
+ static void clearUnownedRoles(Shepherd sh, String username) {
+ javax.jdo.Query> query = sh.getPM().newQuery(Role.class,
+ "username == :username");
+ try {
+ sh.getPM().deletePersistentAll((java.util.Collection>)query.execute(username));
+ } finally { query.closeAll(); }
+ }
+
+ static boolean usernameAvailable(Shepherd sh, String username, String userId) {
+ if (username == null || username.isBlank()) return true;
+ javax.jdo.Query> query = sh.getPM().newQuery(User.class, "username == :username && uuid != :id");
+ try {
+ query.setResult("count(this)");
+ return ((Number)query.execute(username, userId)).longValue() == 0;
+ } finally { query.closeAll(); }
+ }
+
+ static void preserveSubmissionRole(List rolesToReplace, String username, boolean siteAdmin) {
+ rolesToReplace.removeIf(role -> {
+ if (Role.canEditRole(role.getRolename(), siteAdmin)) return false;
+ role.setUsername(username);
+ return true;
+ });
+ }
+
private void addErrorMessage(JSONObject res, String error) {
res.put("error", error);
}
@@ -96,6 +121,20 @@ public void doPost(HttpServletRequest request, HttpServletResponse response)
} else {
newUser = new User(uuid);
}
+ if (username != null) username = username.trim();
+ if (!usernameAvailable(myShepherd, username, uuid)) {
+ response.sendError(HttpServletResponse.SC_CONFLICT);
+ myShepherd.rollbackDBTransaction();
+ return;
+ }
+ if (!request.isUserInRole("admin") && originalUsername != null && newUser.isAdmin(myShepherd)) {
+ response.sendError(HttpServletResponse.SC_FORBIDDEN);
+ myShepherd.rollbackDBTransaction();
+ return;
+ }
+ if (username != null && !username.equals(originalUsername)) {
+ clearUnownedRoles(myShepherd, username);
+ }
if (myShepherd.getUserByUUID(uuid) == null) {
// new User
// System.out.println("hashed password: "+hashedPassword+" with salt "+salt + " from source password "+password);
@@ -190,7 +229,10 @@ public void doPost(HttpServletRequest request, HttpServletResponse response)
List preexistingRoles = new ArrayList();
if (!createThisUser) {
// get existing roles for this existing user
- preexistingRoles = myShepherd.getAllRolesForUser(username);
+ preexistingRoles = originalUsername == null ? new ArrayList()
+ : myShepherd.getAllRolesForUser(originalUsername);
+ // Keep administrator-managed enrollment on unrelated non-admin edits.
+ preserveSubmissionRole(preexistingRoles, newUser.getUsername(), request.isUserInRole("admin"));
if (!preexistingRoles.isEmpty()) permissionsChanged = true;
myShepherd.getPM().deletePersistentAll(preexistingRoles);
}
@@ -206,7 +248,7 @@ public void doPost(HttpServletRequest request, HttpServletResponse response)
// System.out.println("numRoles in context"+d+" is: "+numRoles);
for (int i = 0; i < numRoles; i++) {
String thisRole = roles[i].trim();
- if (!thisRole.trim().equals("")) {
+ if (!thisRole.trim().equals("") && Role.canEditRole(thisRole, request.isUserInRole("admin"))) {
Role role = new Role();
if (myShepherd.getRole(thisRole, username,
("context" + d)) == null) {
diff --git a/src/main/resources/agent-skills/api-reference.md b/src/main/resources/agent-skills/api-reference.md
index 34258673e1..0caff98e55 100644
--- a/src/main/resources/agent-skills/api-reference.md
+++ b/src/main/resources/agent-skills/api-reference.md
@@ -14,7 +14,7 @@ draft/upload/validate/commit lifecycle; the read-only token described here canno
## Security — read first
- **Never ask for, accept, or store the user's Wildbook username or password.** You do not need them.
-- The user generates a short-lived **bearer token** in Wildbook's UI (Account menu → **API Access**)
+- The user generates a short-lived **bearer token** in Wildbook's UI (Account menu → **API Access** → **Read data**)
and pastes **only the token** to you.
- Treat the token as a secret: never log or persist it, never send it anywhere except Wildbook over
HTTPS. It expires after a fixed lifetime that is **configured per Wildbook instance** (commonly
diff --git a/src/main/resources/agent-skills/index.md b/src/main/resources/agent-skills/index.md
index 1436e01eef..760311ca5d 100644
--- a/src/main/resources/agent-skills/index.md
+++ b/src/main/resources/agent-skills/index.md
@@ -7,13 +7,15 @@ exactly what to do; you review and make the final decisions in Wildbook.
## What you'll need
Most tools here need a short-lived access token from Wildbook. In Wildbook, open your account menu
-and choose **API Access** to create one, then paste **only that token** to your assistant — never
+and choose **API Access → Read data** to create one, then paste **only that token** to your assistant — never
your username or password. The token has an expiration date that may vary by Wildbook; create a
fresh one when it stops working. Full technical detail is in the **api-reference** page (fetch
`/api/v3/agent-skill/api-reference`). The import-prep tools below are the exception — they need no
token, because they only prepare files you upload yourself. Direct API submission is a separate,
-limited pilot: its tool needs an enrolled account and a **submissions:write** token supplied by
-your operator. The ordinary API Access token does not grant submission access.
+limited pilot: a site administrator must assign your account the **api-submission** role.
+Then choose **API Access → Data import** and confirm your password to create a
+**submissions:write** token. The **Read data** token does not grant submission access,
+and the **Data import** token does not work with the general read API.
## Check and tidy your catalog (read-only — the tools only suggest; you make the changes in Wildbook)
diff --git a/src/main/resources/agent-skills/submit-sightings.md b/src/main/resources/agent-skills/submit-sightings.md
index 47ee4a465c..e2cf8bf214 100644
--- a/src/main/resources/agent-skills/submit-sightings.md
+++ b/src/main/resources/agent-skills/submit-sightings.md
@@ -26,9 +26,14 @@ assign an individual identity. Existing records are not updated.
- The installation's exact base URL, including any application prefix. For example,
`https://example.org/wildbook` means the API is below `/wildbook/api/v3`.
-- An account enrolled by the installation operator and a short-lived
- `submissions:write` bearer token. A normal API Access/search token will not work.
- The operator obtains the scoped token using a trusted client with fresh HTTP Basic
+- An account explicitly granted the **api-submission** role by a site administrator
+ in this installation's context0 user editor, and a short-lived `submissions:write`
+ bearer token. The account owner can open **API Access**, select **Data import**
+ under **Token purpose**, click **Generate API token**, and confirm their password.
+ The default **Read data** token does not work with submissions. A Data import
+ token does not work with the general read API; request a separate Read data token
+ if your workflow also searches existing sightings.
+ A trusted client can alternatively obtain the scoped token using fresh HTTP Basic
credentials **for the enrolled account that should own the imported records**, at
`POST /api/v3/auth/token?scope=submissions:write`. Use a non-admin integration
account for the pilot; do not mint with an operator's own account merely because
@@ -38,6 +43,9 @@ assign an individual identity. Existing records are not updated.
only the token through the runtime's secret mechanism; do not request their password.
The response's `expiresInSeconds` is authoritative. `submissions:read` permits
status/results reads, including after write enrollment is removed.
+ Removing the role blocks new writes with existing tokens and prevents unstarted
+ imports from executing. It does not cancel an import already executing or remove
+ imported records. Status access remains subject to ownership and token validity.
- Local JPEG or PNG files you are authorized to import, sighting dates and species,
and the correct configured Wildbook location IDs. Do not invent missing facts.
- Durable local job state: original create body/key, submission ID, latest revision,
@@ -425,8 +433,8 @@ refer those phases to the operator.
| Lost row-replacement response | GET `/rows` and compare the complete intended rows. Equivalent JSON numbers such as 2025 and 2025.0 may serialize differently; compare values while keeping booleans distinct. Do not overwrite unexplained edits. |
| Lost validation response | If the draft is still editable and no commit intent is pending, repeat validation with the current ETag. Validation does not increment revision, but each run creates a new report ID; only the latest report can be committed. |
| Lost commit response | GET the submission first. If an operation ID exists, poll that accepted operation. Otherwise retry only the saved commit body/key/revision; do not generate a new key or silently revalidate a frozen intent. |
-| HTTP 401 | Token may be expired/invalid or have the wrong audience. An ordinary API Access/search token also gets 401, including on the first request. Obtain a token explicitly minted with `scope=submissions:write` for the intended owner (or `submissions:read` for reads); do not keep regenerating ordinary search tokens. Resume with saved job state. |
-| HTTP 403 | A valid submissions read token was used for a write, or enrollment/access is unavailable. Contact the operator; cookies do not substitute for scoped tokens. |
+| HTTP 401 | Token may be expired/invalid or have the wrong audience. An API Access **Read data** token also gets 401, including on the first request. Choose **API Access → Data import**, or obtain a token explicitly minted with `scope=submissions:write` for the intended owner (or `submissions:read` for reads); do not keep regenerating ordinary search tokens. Resume with saved job state. |
+| HTTP 403 | A valid submissions read token was used for a write, or enrollment/access is unavailable. `ACCESS_DENIED` with `Account requires the api-submission role` means a site administrator must grant that role to the intended owner. Contact the operator; cookies do not substitute for scoped tokens. |
| HTTP 404 | Verify base URL, deployment and saved ID, and check content type. For a non-admin account, a different owner's submission is hidden as not found; do not probe other IDs. |
| HTTP 400 `BAD_REQUEST` | Check JSON/envelope and filename rules. If-Match must be a quoted numeric revision, not unquoted `3` or weak `W/"3"`. Correct the request rather than blindly retrying. |
| HTTP 428 | Supply the current quoted If-Match ETag. |
diff --git a/src/main/resources/bundles/apiAccessKeys.properties b/src/main/resources/bundles/apiAccessKeys.properties
index 0ad5f78453..d8cf6ddd49 100644
--- a/src/main/resources/bundles/apiAccessKeys.properties
+++ b/src/main/resources/bundles/apiAccessKeys.properties
@@ -34,8 +34,9 @@
# Set real values only in the private data-dir override described above.
# Allow new submissions and edits for enrolled integration accounts.
#submissions.enabled = false
-# Comma-separated Wildbook user UUIDs; an empty/unset list enrolls nobody.
-#submissions.allowedUserIds =
+# Enroll users by assigning api-submission in the site administrator user editor (context0).
+# The former submissions.allowedUserIds setting is no longer used.
+# Removing the role blocks writes and imports that have not started; status reads remain available.
# Private staging path INSIDE the container, matching the deployment Compose mount.
# Pre-create the host directory with service UID/GID ownership and permissions 0700.
# Must not overlap webapps, legacy uploads, imports, or any local asset-store root.
diff --git a/src/main/webapp/appadmin/users.jsp b/src/main/webapp/appadmin/users.jsp
index 683dd3428b..73f215a822 100755
--- a/src/main/webapp/appadmin/users.jsp
+++ b/src/main/webapp/appadmin/users.jsp
@@ -31,7 +31,8 @@ String localEmail="";
Shepherd myShepherd = new Shepherd(context);
myShepherd.setAction("users.jsp");
-List roles=CommonConfiguration.getIndexedPropertyValues("role",context);
+List roles=new ArrayList(CommonConfiguration.getIndexedPropertyValues("role",context));
+if (!roles.contains(Role.API_SUBMISSION)) roles.add(Role.API_SUBMISSION);
List roleDefinitions=CommonConfiguration.getIndexedPropertyValues("roleDefinition",context);
int numRoles=roles.size();
int numRoleDefinitions=roleDefinitions.size();
@@ -692,7 +693,9 @@ try {
}
//now one last check: only let someone who has a role assign the role
- if(request.isUserInRole("admin") || request.isUserInRole(roles.get(q))){
+ if((d == 0 || !Role.API_SUBMISSION.equals(roles.get(q))) &&
+ Role.canEditRole(roles.get(q), request.isUserInRole("admin")) &&
+ (request.isUserInRole("admin") || request.isUserInRole(roles.get(q)))){
%><%
}
}%>
diff --git a/src/test/java/org/ecocean/RoleTest.java b/src/test/java/org/ecocean/RoleTest.java
index 36cfeaf902..f01cae0440 100644
--- a/src/test/java/org/ecocean/RoleTest.java
+++ b/src/test/java/org/ecocean/RoleTest.java
@@ -19,10 +19,13 @@ class RoleTest {
Role.SYSTEM_ROLES, "UserConsolidate walks this order to rank two users");
}
- @Test void systemRoleNamesHoldsExactlyTheSameNames() {
- assertEquals(new LinkedHashSet(Role.SYSTEM_ROLES), Role.SYSTEM_ROLE_NAMES);
+ @Test void systemRoleNamesAlsoReserveExplicitCapabilities() {
+ LinkedHashSet expected = new LinkedHashSet(Role.SYSTEM_ROLES);
+ expected.add(Role.API_SUBMISSION);
+ assertEquals(expected, Role.SYSTEM_ROLE_NAMES);
+ org.junit.jupiter.api.Assertions.assertFalse(Role.SYSTEM_ROLES.contains(Role.API_SUBMISSION));
assertTrue(Role.SYSTEM_ROLE_NAMES.contains("orgAdmin"));
- assertEquals(Role.SYSTEM_ROLES.size(), Role.SYSTEM_ROLE_NAMES.size(), "no duplicates");
+ assertEquals(Role.SYSTEM_ROLES.size() + 1, Role.SYSTEM_ROLE_NAMES.size(), "no duplicates");
}
@Test void systemRolesAreImmutable() {
diff --git a/src/test/java/org/ecocean/api/AuthTokenSubmissionScopeTest.java b/src/test/java/org/ecocean/api/AuthTokenSubmissionScopeTest.java
index a0ecc0df05..fb05947d1f 100644
--- a/src/test/java/org/ecocean/api/AuthTokenSubmissionScopeTest.java
+++ b/src/test/java/org/ecocean/api/AuthTokenSubmissionScopeTest.java
@@ -23,15 +23,18 @@ private void request(String scope, boolean enabled, boolean enrolled, int expect
when(request.getParameter("scope")).thenReturn(scope);
HttpServletResponse response = mock(HttpServletResponse.class);
StringWriter output = new StringWriter(); when(response.getWriter()).thenReturn(new PrintWriter(output));
- User user = mock(User.class); when(user.checkPassword("password")).thenReturn(true); when(user.getId()).thenReturn("pilot-id");
+ User user = mock(User.class); when(user.checkPassword("password")).thenReturn(true); when(user.getId()).thenReturn("pilot-id"); when(user.getUsername()).thenReturn("pilot");
+ javax.jdo.Query query = mock(javax.jdo.Query.class);
+ when(query.execute("pilot", org.ecocean.Role.API_SUBMISSION, "context0")).thenReturn(enrolled ? 1L : 0L);
+ javax.jdo.PersistenceManager pm = mock(javax.jdo.PersistenceManager.class);
+ when(pm.newQuery(eq(org.ecocean.Role.class), anyString())).thenReturn(query);
JwtService jwt = mock(JwtService.class); when(jwt.isEnabled()).thenReturn(true);
when(jwt.signSubmission(anyString(), anyString(), anyLong(), anyString())).thenReturn("submission-token");
- try (MockedConstruction sh = mockConstruction(Shepherd.class, (m,c) -> when(m.getUser("pilot")).thenReturn(user));
+ try (MockedConstruction sh = mockConstruction(Shepherd.class, (m,c) -> { when(m.getUser("pilot")).thenReturn(user); when(m.getUserByUUID("pilot-id")).thenReturn(user); when(m.getPM()).thenReturn(pm); when(m.getContext()).thenReturn("context0"); });
MockedStatic config = mockStatic(CommonConfiguration.class);
MockedStatic js = mockStatic(JwtService.class)) {
config.when(() -> CommonConfiguration.getApiAccessProperty("submissions.enabled", "context0")).thenReturn(Boolean.toString(enabled));
- config.when(() -> CommonConfiguration.getApiAccessProperty("submissions.allowedUserIds", "context0"))
- .thenReturn(enrolled ? "other, pilot-id" : "other");
+
js.when(() -> JwtService.fromConfig("context0")).thenReturn(jwt);
new AuthToken().doPost(request, response);
verify(response).setStatus(expected);
diff --git a/src/test/java/org/ecocean/api/submission/SubmissionPolicyTest.java b/src/test/java/org/ecocean/api/submission/SubmissionPolicyTest.java
index 883ac046d2..0a0c983056 100644
--- a/src/test/java/org/ecocean/api/submission/SubmissionPolicyTest.java
+++ b/src/test/java/org/ecocean/api/submission/SubmissionPolicyTest.java
@@ -19,4 +19,36 @@ class SubmissionPolicyTest {
config.verify(() -> CommonConfiguration.getProperty("submissions.enabled", "context0"), never());
}
}
+
+ @Test void persistedRoleRevocationTakesEffectAndLegacyAllowlistCannotRestoreAccess() {
+ org.ecocean.User user = mock(org.ecocean.User.class);
+ when(user.getUsername()).thenReturn("pilot");
+ javax.jdo.Query query = mock(javax.jdo.Query.class);
+ when(query.execute("pilot", org.ecocean.Role.API_SUBMISSION, "context0")).thenReturn(1L, 0L);
+ javax.jdo.PersistenceManager pm = mock(javax.jdo.PersistenceManager.class);
+ when(pm.newQuery(eq(org.ecocean.Role.class), anyString())).thenReturn(query);
+ try (org.mockito.MockedConstruction shepherds =
+ mockConstruction(org.ecocean.shepherd.core.Shepherd.class, (sh, c) -> {
+ when(sh.getUserByUUID("pilot-id")).thenReturn(user);
+ when(sh.getPM()).thenReturn(pm); when(sh.getContext()).thenReturn("context0");
+ });
+ MockedStatic config = mockStatic(CommonConfiguration.class)) {
+ config.when(() -> CommonConfiguration.getApiAccessProperty("submissions.enabled", "context0")).thenReturn("true");
+ config.when(() -> CommonConfiguration.getApiAccessProperty("submissions.allowedUserIds", "context0")).thenReturn("pilot-id");
+ SubmissionPolicy.requireAdmission("context0", "pilot-id");
+ SubmissionException denied = assertThrows(SubmissionException.class,
+ () -> SubmissionPolicy.requireAdmission("context0", "pilot-id"));
+ assertEquals(403, denied.status);
+ for (org.ecocean.shepherd.core.Shepherd sh : shepherds.constructed()) verify(sh).rollbackAndClose();
+ verify(query, times(2)).setIgnoreCache(true);
+ verify(query, times(2)).closeAll();
+ }
+ }
+ @Test void missingAccountIsNotEnrolledAndClosesTransaction() {
+ try (org.mockito.MockedConstruction shepherds =
+ mockConstruction(org.ecocean.shepherd.core.Shepherd.class)) {
+ assertFalse(SubmissionPolicy.enrolled("context0", "missing"));
+ verify(shepherds.constructed().get(0)).rollbackAndClose();
+ }
+ }
}
diff --git a/src/test/java/org/ecocean/api/submission/SubmissionStoreDbTest.java b/src/test/java/org/ecocean/api/submission/SubmissionStoreDbTest.java
index 0f9e1a613a..b0b08788e4 100644
--- a/src/test/java/org/ecocean/api/submission/SubmissionStoreDbTest.java
+++ b/src/test/java/org/ecocean/api/submission/SubmissionStoreDbTest.java
@@ -44,6 +44,39 @@ class SubmissionStoreDbTest {
private JSONObject rows(int year) { return new JSONObject().put("rows", new JSONArray().put(new JSONObject()
.put("clientRowId", "row-1").put("fields", new JSONObject().put("Encounter.year", year)))); }
+ private boolean persistedEnrollment(String id) {
+ Shepherd sh = new Shepherd("context0", properties);
+ try {
+ sh.beginDBTransaction();
+ return SubmissionPolicy.enrolled(sh, id);
+ } finally { sh.rollbackAndClose(); }
+ }
+
+ @Test void enrollmentTracksPersistedRoleGrantAndRevocation() {
+ String id = UUID.randomUUID().toString();
+ String username = "pilot'" + id;
+ Shepherd sh = new Shepherd("context0", properties);
+ try {
+ sh.beginDBTransaction();
+ org.ecocean.User user = new org.ecocean.User(username, id);
+ user.setUsername(username);
+ sh.getPM().makePersistent(user);
+ org.ecocean.Role role = new org.ecocean.Role(username, org.ecocean.Role.API_SUBMISSION);
+ role.setContext("context1");
+ sh.getPM().makePersistent(role);
+ assertTrue(sh.commitDBTransactionWithStatus());
+ assertFalse(persistedEnrollment(id));
+ sh.beginDBTransaction();
+ role.setContext("context0");
+ assertTrue(sh.commitDBTransactionWithStatus());
+ assertTrue(persistedEnrollment(id));
+ sh.beginDBTransaction();
+ sh.getPM().deletePersistent(role);
+ assertTrue(sh.commitDBTransactionWithStatus());
+ assertFalse(persistedEnrollment(id));
+ } finally { sh.rollbackAndClose(); }
+ }
+
@Test void durableRowsOwnershipAndOriginalReplay() {
String owner = UUID.randomUUID().toString();
JSONObject first = store.create("context0", owner, "key", create()); String id = first.getString("id");
diff --git a/src/test/java/org/ecocean/security/LocationRoleAccessTest.java b/src/test/java/org/ecocean/security/LocationRoleAccessTest.java
index 19161fab36..791dc5e44b 100644
--- a/src/test/java/org/ecocean/security/LocationRoleAccessTest.java
+++ b/src/test/java/org/ecocean/security/LocationRoleAccessTest.java
@@ -94,6 +94,7 @@ private static Set set(String... names) {
"a location named exactly like a system role grants nothing");
assertTrue(LocationRoleAccess.roleNamesFor("admin").isEmpty());
assertTrue(LocationRoleAccess.roleNamesFor("orgAdmin").isEmpty());
+ assertTrue(LocationRoleAccess.roleNamesFor("api-submission").isEmpty());
}
// pure traversal edge cases live with the traversal, in LocationIDLineageTest
diff --git a/src/test/java/org/ecocean/servlet/UserSubmissionRoleDbTest.java b/src/test/java/org/ecocean/servlet/UserSubmissionRoleDbTest.java
new file mode 100644
index 0000000000..8bb98c6b66
--- /dev/null
+++ b/src/test/java/org/ecocean/servlet/UserSubmissionRoleDbTest.java
@@ -0,0 +1,50 @@
+package org.ecocean.servlet;
+
+import java.util.Properties;
+import java.util.UUID;
+import org.ecocean.Role;
+import org.ecocean.User;
+import org.ecocean.shepherd.core.Shepherd;
+import org.ecocean.shepherd.core.TestPMFUtil;
+import org.junit.jupiter.api.Test;
+import org.testcontainers.containers.PostgreSQLContainer;
+import org.testcontainers.junit.jupiter.Container;
+import org.testcontainers.junit.jupiter.Testcontainers;
+import static org.junit.jupiter.api.Assertions.*;
+
+@Testcontainers
+class UserSubmissionRoleDbTest {
+ @Container static PostgreSQLContainer> postgres = new PostgreSQLContainer<>("postgres:15-alpine");
+
+ @Test void persistedUsernameOwnershipAndPriorRoleCleanup() {
+ TestPMFUtil.closePMF("context0");
+ Properties properties = new Properties();
+ properties.setProperty("datanucleus.ConnectionUserName", postgres.getUsername());
+ properties.setProperty("datanucleus.ConnectionPassword", postgres.getPassword());
+ properties.setProperty("datanucleus.ConnectionDriverName", postgres.getDriverClassName());
+ properties.setProperty("datanucleus.ConnectionURL", postgres.getJdbcUrl());
+ properties.setProperty("datanucleus.schema.autoCreateAll", "true");
+ Shepherd sh = new Shepherd("context0", properties);
+ try {
+ sh.beginDBTransaction();
+ String id = UUID.randomUUID().toString();
+ User user = new User("test@example.invalid", id); user.setUsername("current");
+ sh.getPM().makePersistent(user);
+ Role priorAdmin = new Role("unused", "admin"); priorAdmin.setContext("context0");
+ Role priorImport = new Role("unused", Role.API_SUBMISSION); priorImport.setContext("context0");
+ sh.getPM().makePersistent(priorAdmin); sh.getPM().makePersistent(priorImport);
+ assertTrue(sh.commitDBTransactionWithStatus());
+ sh.beginDBTransaction();
+ assertTrue(UserCreate.usernameAvailable(sh, "current", id));
+ assertFalse(UserCreate.usernameAvailable(sh, "current", "another-account"));
+ assertTrue(UserCreate.usernameAvailable(sh, "unused", id));
+ UserCreate.clearUnownedRoles(sh, "unused");
+ assertTrue(sh.commitDBTransactionWithStatus());
+ sh.beginDBTransaction();
+ assertTrue(sh.getAllRolesForUser("unused").isEmpty());
+ } finally {
+ sh.rollbackAndClose();
+ TestPMFUtil.closePMF("context0");
+ }
+ }
+}
diff --git a/src/test/java/org/ecocean/servlet/UserSubmissionRoleTest.java b/src/test/java/org/ecocean/servlet/UserSubmissionRoleTest.java
new file mode 100644
index 0000000000..04cc12cae9
--- /dev/null
+++ b/src/test/java/org/ecocean/servlet/UserSubmissionRoleTest.java
@@ -0,0 +1,67 @@
+package org.ecocean.servlet;
+
+import java.util.ArrayList;
+import java.util.List;
+import org.ecocean.Role;
+import org.junit.jupiter.api.Test;
+import static org.junit.jupiter.api.Assertions.*;
+import static org.mockito.Mockito.*;
+
+class UserSubmissionRoleTest {
+ @Test void onlySiteAdminCanChangeEnrollment() {
+ assertFalse(Role.canEditRole(Role.API_SUBMISSION, false));
+ assertTrue(Role.canEditRole(Role.API_SUBMISSION, true));
+ }
+ @Test void nonAdminEditPreservesEnrollmentAcrossRename() {
+ Role capability = new Role("old-name", Role.API_SUBMISSION);
+ List toReplace = new ArrayList<>(List.of(capability, new Role("old-name", "researcher")));
+ UserCreate.preserveSubmissionRole(toReplace, "new-name", false);
+ assertFalse(toReplace.contains(capability));
+ assertEquals("new-name", capability.getUsername());
+ assertEquals(1, toReplace.size());
+ }
+ @Test void adminCanRevokeEnrollmentByOmittingItFromReplacement() {
+ Role capability = new Role("pilot", Role.API_SUBMISSION);
+ List toReplace = new ArrayList<>(List.of(capability));
+ UserCreate.preserveSubmissionRole(toReplace, "pilot", true);
+ assertTrue(toReplace.contains(capability));
+ }
+ @Test void usernameCollisionIsRejectedAndQueryClosed() {
+ org.ecocean.shepherd.core.Shepherd sh = mock(org.ecocean.shepherd.core.Shepherd.class);
+ javax.jdo.PersistenceManager pm = mock(javax.jdo.PersistenceManager.class);
+ javax.jdo.Query query = mock(javax.jdo.Query.class);
+ when(sh.getPM()).thenReturn(pm);
+ when(pm.newQuery(eq(org.ecocean.User.class), anyString())).thenReturn(query);
+ when(query.execute("taken", "account-a")).thenReturn(1L);
+ assertFalse(UserCreate.usernameAvailable(sh, "taken", "account-a"));
+ verify(query).closeAll();
+ }
+
+ @Test void accountMergeDoesNotTransferSubmissionEnrollment() {
+ org.ecocean.shepherd.core.Shepherd sh = mock(org.ecocean.shepherd.core.Shepherd.class);
+ javax.jdo.PersistenceManager pm = mock(javax.jdo.PersistenceManager.class);
+ when(sh.getPM()).thenReturn(pm);
+ when(sh.getContext()).thenReturn("context0");
+ org.ecocean.User retained = mock(org.ecocean.User.class), merged = mock(org.ecocean.User.class);
+ when(retained.getUsername()).thenReturn("retained"); when(merged.getUsername()).thenReturn("merged");
+ Role role = new Role("merged", Role.API_SUBMISSION);
+ when(sh.getAllRolesForUserInContext("merged", "context0")).thenReturn(List.of(role));
+ when(sh.getAllRolesForUserInContext("retained", "context0")).thenReturn(List.of());
+ UserConsolidate.consolidateRoles(sh, retained, merged);
+ verify(pm).deletePersistent(role);
+ verify(pm, never()).makePersistent(role);
+ assertEquals("merged", role.getUsername());
+ }
+ @Test void adoptingUnusedUsernameClearsAllPriorGrants() {
+ org.ecocean.shepherd.core.Shepherd sh = mock(org.ecocean.shepherd.core.Shepherd.class);
+ javax.jdo.PersistenceManager pm = mock(javax.jdo.PersistenceManager.class);
+ javax.jdo.Query query = mock(javax.jdo.Query.class);
+ when(sh.getPM()).thenReturn(pm);
+ when(pm.newQuery(eq(Role.class), anyString())).thenReturn(query);
+ List prior = List.of(new Role("reused", Role.API_SUBMISSION), new Role("reused", "admin"));
+ when(query.execute("reused")).thenReturn(prior);
+ UserCreate.clearUnownedRoles(sh, "reused");
+ verify(pm).deletePersistentAll(prior);
+ verify(query).closeAll();
+ }
+}
From c73c34d1ef227d7dcda8572bcbd794c1c406dc83 Mon Sep 17 00:00:00 2001
From: JasonWildMe
Date: Fri, 25 Sep 2026 12:48:47 -0700
Subject: [PATCH 6/7] Judge future dates by the earliest time zone, not the
server's
dateIsInFuture compared submitted dates with the server JVM's local date,
so a submitter who was already a calendar day ahead of the server (for
example Australia or New Zealand against a UTC or U.S. Pacific host) had
their legitimate same-day observation rejected as "in the future". On a
UTC server this rejects Sydney's "today" for 10 hours a day; on a Pacific
server, 17 hours. The same edge rejected New Year's Day while the server
was still on December 31.
Compare against the current date at UTC+14, the earliest civil calendar
date on Earth, so any observer's local current date is accepted while
genuinely future dates are still rejected. Instant-based checks
(dateInMilliseconds) are unaffected. The shared helper also backs the
legacy bulk importer, EncounterForm, EncounterPatchValidator and
Encounter date setters, which all gain the same tolerance.
An overload taking an explicit "today" makes the tests deterministic,
replacing the previous test that depended on the server's local date.
Reviewed by Codex; its test-determinism finding is addressed.
Co-Authored-By: Claude Opus 5.5
Co-Authored-By: Codex
---
src/main/java/org/ecocean/Util.java | 20 +++++--
.../agent-skills/submit-sightings.md | 2 +-
src/test/java/org/ecocean/UtilTest.java | 55 +++++++++++++------
3 files changed, 55 insertions(+), 22 deletions(-)
diff --git a/src/main/java/org/ecocean/Util.java b/src/main/java/org/ecocean/Util.java
index 5ca8e58e61..02464aa9ca 100644
--- a/src/main/java/org/ecocean/Util.java
+++ b/src/main/java/org/ecocean/Util.java
@@ -2,7 +2,6 @@
import java.util.ArrayList;
import java.util.Arrays;
-import java.util.Calendar;
import java.util.Collection;
import java.util.Collections;
import java.util.Enumeration;
@@ -12,6 +11,8 @@
import java.util.UUID;
import java.text.SimpleDateFormat;
+import java.time.LocalDate;
+import java.time.ZoneOffset;
import java.util.Date;
import org.ecocean.media.AssetStore;
@@ -804,12 +805,21 @@ public static boolean dateTimeIsOnlyDate(DateTime dt) {
}
}
+ // UTC+14 (Line Islands) is the earliest civil calendar date on Earth. Comparing against it,
+ // rather than the server's own zone, avoids rejecting a submitter's legitimate "today" when
+ // they are already a calendar day ahead of the server (e.g. Australia vs a UTC/Pacific host).
+ public static final ZoneOffset LATEST_CIVIL_OFFSET = ZoneOffset.ofHours(14);
+
public static boolean dateIsInFuture(Integer year, Integer month, Integer day) {
+ return dateIsInFuture(year, month, day, LocalDate.now(LATEST_CIVIL_OFFSET));
+ }
+
+ // (partial) date is future only if it is later than the given "today" at its own precision
+ public static boolean dateIsInFuture(Integer year, Integer month, Integer day, LocalDate today) {
if (year == null) return false;
- Calendar cal = Calendar.getInstance();
- int nowY = cal.get(Calendar.YEAR);
- int nowM = cal.get(Calendar.MONTH) + 1; // frikken zero-based months!
- int nowD = cal.get(Calendar.DAY_OF_MONTH);
+ int nowY = today.getYear();
+ int nowM = today.getMonthValue();
+ int nowD = today.getDayOfMonth();
if (year > nowY) return true;
if (month == null) return false; // only have year
if ((year == nowY) && (month > nowM)) return true;
diff --git a/src/main/resources/agent-skills/submit-sightings.md b/src/main/resources/agent-skills/submit-sightings.md
index e2cf8bf214..b1aaef5009 100644
--- a/src/main/resources/agent-skills/submit-sightings.md
+++ b/src/main/resources/agent-skills/submit-sightings.md
@@ -173,7 +173,7 @@ Supported fields for this pilot are exactly:
|---|---|---|
| `Encounter.genus` | string | Required. Scientific genus; combined with specific epithet must match a configured taxonomy. |
| `Encounter.specificEpithet` | string | Required. Configured scientific-name suffix after the genus, including a subspecies word if present; not the full name or common name. |
-| `Encounter.year` | integer | Required, at least 1000; the represented date must not be in the future. |
+| `Encounter.year` | integer | Required, at least 1000; the represented date must not be in the future. "Today" is judged by the most advanced civil time zone (UTC+14), so the observer's local current date is accepted. |
| `Encounter.month` | integer | Optional, 1–12. Required when day is supplied. |
| `Encounter.day` | integer | Optional; must exist in the supplied year/month, including leap-year rules. |
| `Encounter.hour` | integer | Optional, 0–23. Supply only a known observation time; no timezone field is supported here. |
diff --git a/src/test/java/org/ecocean/UtilTest.java b/src/test/java/org/ecocean/UtilTest.java
index 70999ad8e4..6ddf40f77b 100644
--- a/src/test/java/org/ecocean/UtilTest.java
+++ b/src/test/java/org/ecocean/UtilTest.java
@@ -1,6 +1,8 @@
package org.ecocean;
-import java.util.Calendar;
+import java.time.Instant;
+import java.time.LocalDate;
+import java.time.ZoneOffset;
import java.util.List;
import org.junit.jupiter.api.Test;
import static org.junit.Assert.*;
@@ -27,22 +29,43 @@ class UtilTest {
assertEquals(testVal, Util.roundISO8601toMillis(testVal));
}
- // note there is an extremely slim chance that if this test is run a couple cpu
- // cycles before midnight, it might return invalid results. taking my chances.
@Test void testDateFuture() {
- Calendar cal = Calendar.getInstance();
- int year = cal.get(Calendar.YEAR);
- int month = cal.get(Calendar.MONTH) + 1; // frikken zero-based months!
- int day = cal.get(Calendar.DAY_OF_MONTH);
-
- assertFalse(Util.dateIsInFuture(null, null, null));
- assertFalse(Util.dateIsInFuture(year - 1, null, null));
- assertFalse(Util.dateIsInFuture(year, null, null));
- assertFalse(Util.dateIsInFuture(year, month, null));
- assertFalse(Util.dateIsInFuture(year, month, day));
- assertTrue(Util.dateIsInFuture(year, month + 1, null));
- assertTrue(Util.dateIsInFuture(year, month, day + 1));
- assertTrue(Util.dateIsInFuture(year + 1, month, day));
+ LocalDate today = LocalDate.of(2026, 9, 25);
+
+ assertFalse(Util.dateIsInFuture(null, null, null, today));
+ assertFalse(Util.dateIsInFuture(2025, null, null, today));
+ assertFalse(Util.dateIsInFuture(2026, null, null, today));
+ assertFalse(Util.dateIsInFuture(2026, 9, null, today));
+ assertFalse(Util.dateIsInFuture(2026, 9, 25, today));
+ assertFalse(Util.dateIsInFuture(2026, 8, 31, today));
+ assertTrue(Util.dateIsInFuture(2026, 10, null, today));
+ assertTrue(Util.dateIsInFuture(2026, 9, 26, today));
+ assertTrue(Util.dateIsInFuture(2027, null, null, today));
+ assertTrue(Util.dateIsInFuture(2027, 1, 1, today));
+ // year boundary: the next calendar year is future only once it has begun
+ LocalDate newYearsEve = LocalDate.of(2026, 12, 31);
+ assertFalse(Util.dateIsInFuture(2026, 12, 31, newYearsEve));
+ assertTrue(Util.dateIsInFuture(2027, 1, 1, newYearsEve));
+ }
+
+ // at 19:00 UTC a UTC server is still on the 25th while Sydney (UTC+10) is already on the 26th;
+ // the observer's "today" must not be rejected, but a date beyond UTC+14's today still is
+ @Test void testDateFutureAllowsSubmittersAheadOfServer() {
+ Instant now = Instant.parse("2026-09-25T19:00:00Z");
+ LocalDate latestToday = now.atOffset(Util.LATEST_CIVIL_OFFSET).toLocalDate();
+ LocalDate serverToday = now.atOffset(ZoneOffset.UTC).toLocalDate();
+ LocalDate sydneyToday = now.atOffset(ZoneOffset.ofHours(10)).toLocalDate();
+
+ assertEquals(LocalDate.of(2026, 9, 25), serverToday);
+ assertEquals(LocalDate.of(2026, 9, 26), sydneyToday);
+ assertFalse(Util.dateIsInFuture(2026, 9, 26, latestToday));
+ assertFalse(Util.dateIsInFuture(2026, 9, 25, latestToday));
+ assertTrue(Util.dateIsInFuture(2026, 9, 27, latestToday));
+ // the clock-based entry point accepts the current date at UTC+14 (a later read can only
+ // move "today" forward, so this cannot flake at midnight)
+ LocalDate latestNow = LocalDate.now(Util.LATEST_CIVIL_OFFSET);
+ assertFalse(Util.dateIsInFuture(latestNow.getYear(), latestNow.getMonthValue(),
+ latestNow.getDayOfMonth()));
}
@Test void testHumanApprox() {
From b1408b2b5d2e071507122efd3ea85a03a59c863c Mon Sep 17 00:00:00 2001
From: JasonWildMe
Date: Fri, 25 Sep 2026 14:23:27 -0700
Subject: [PATCH 7/7] Report why a submission field was rejected, on the field
that caused it
Validation reports turned every legacy bulk-import rejection into
INVALID_VALUE "Value failed bulk-import validation". A future day or
month was reported on Encounter.year, parse failures leaked Java
exception text, and an unconfigured locationID produced two issues. An
agent following the skill could only guess, and tended to "correct" a
year that was right.
INVALID_VALUE issues now carry an optional, additive reason (REQUIRED,
REQUIRES_FIELD, UNPARSEABLE, OUT_OF_RANGE, FUTURE_DATE, NOT_CONFIGURED,
INVALID) and a specific message, for example "'F' is not a configured
sex value; use one of: unknown, male, female" or "2025-02 has no day
29". Where legacy validation replaced a supplied value's own error with
a generic "required value", the original cause is recovered. A future
date is reported on the year, month or day that is actually too late,
judged against the UTC+14 date captured before legacy validation runs.
The duplicate legacy location issue is suppressed only when
INVALID_LOCATION was already reported for that row.
Acceptance is unchanged: each legacy rejection still yields exactly one
issue, and valid, normalizedRows and the digests the importer re-checks
are untouched. Legacy bulk import messages are not modified.
The skill documents reasons and says to check the source observation
rather than change values to pass. Both OpenAPI copies describe reason
as an open list, the contract checker asserts parity with the Java
list, and tests run the real legacy validators so wording drift fails.
Plan and code reviewed by Codex; its findings on recovering
overwritten causes, future-date attribution (including a nonexistent day
that is also in the future), bounded allowed-value lists and a fixed test
clock with report-invariant checks are addressed.
Co-Authored-By: Claude Opus 5.5
Co-Authored-By: Codex
---
docs/design/submissions/examples.json | 71 +++++++
docs/design/submissions/openapi.yaml | 15 ++
scripts/submissions/check_contract.py | 6 +
.../api/submission/SubmissionValidator.java | 20 +-
.../api/submission/SubmissionValueIssues.java | 140 +++++++++++++
.../agent-skills/submit-sightings.md | 41 +++-
src/main/resources/openapi.yaml | 15 ++
.../submission/SubmissionValueIssuesTest.java | 185 ++++++++++++++++++
8 files changed, 479 insertions(+), 14 deletions(-)
create mode 100644 src/main/java/org/ecocean/api/submission/SubmissionValueIssues.java
create mode 100644 src/test/java/org/ecocean/api/submission/SubmissionValueIssuesTest.java
diff --git a/docs/design/submissions/examples.json b/docs/design/submissions/examples.json
index d80b7d9f34..922b7663e1 100644
--- a/docs/design/submissions/examples.json
+++ b/docs/design/submissions/examples.json
@@ -90,6 +90,77 @@
}
}
},
+ "validationFailedReasons": {
+ "schema": "Validation",
+ "value": {
+ "id": "00000000-0000-4000-8000-000000000004",
+ "submissionId": "00000000-0000-4000-8000-000000000001",
+ "revision": 2,
+ "valid": false,
+ "configDigest": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
+ "manifestDigest": "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb",
+ "errors": [
+ {
+ "code": "INVALID_VALUE",
+ "reason": "FUTURE_DATE",
+ "message": "2099 is later than today's date in every time zone",
+ "clientRowId": "observation-1",
+ "rowIndex": 0,
+ "field": "Encounter.year"
+ },
+ {
+ "code": "INVALID_VALUE",
+ "reason": "NOT_CONFIGURED",
+ "message": "'F' is not a configured sex value; use one of: unknown, male, female",
+ "clientRowId": "observation-1",
+ "rowIndex": 0,
+ "field": "Encounter.sex"
+ }
+ ],
+ "warnings": [],
+ "normalizedRows": [],
+ "effectiveOwnerId": "00000000-0000-4000-8000-000000000005",
+ "processing": {
+ "mode": "import-only"
+ }
+ }
+ },
+ "rejectEmptyReason": {
+ "schema": "Validation",
+ "value": {
+ "id": "00000000-0000-4000-8000-000000000004",
+ "submissionId": "00000000-0000-4000-8000-000000000001",
+ "revision": 2,
+ "valid": false,
+ "configDigest": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
+ "manifestDigest": "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb",
+ "errors": [
+ {
+ "code": "INVALID_VALUE",
+ "reason": "",
+ "message": "2099 is later than today's date in every time zone",
+ "clientRowId": "observation-1",
+ "rowIndex": 0,
+ "field": "Encounter.year"
+ },
+ {
+ "code": "INVALID_VALUE",
+ "reason": "NOT_CONFIGURED",
+ "message": "'F' is not a configured sex value; use one of: unknown, male, female",
+ "clientRowId": "observation-1",
+ "rowIndex": 0,
+ "field": "Encounter.sex"
+ }
+ ],
+ "warnings": [],
+ "normalizedRows": [],
+ "effectiveOwnerId": "00000000-0000-4000-8000-000000000005",
+ "processing": {
+ "mode": "import-only"
+ }
+ },
+ "valid": false
+ },
"importedResults": {
"schema": "Results",
"value": {
diff --git a/docs/design/submissions/openapi.yaml b/docs/design/submissions/openapi.yaml
index f1694df1fc..1ea9ee475e 100644
--- a/docs/design/submissions/openapi.yaml
+++ b/docs/design/submissions/openapi.yaml
@@ -1263,6 +1263,21 @@ components:
field:
type: string
minLength: 1
+ reason:
+ type: string
+ minLength: 1
+ description: >-
+ Present on INVALID_VALUE issues: why the value was rejected. One of REQUIRED,
+ REQUIRES_FIELD, UNPARSEABLE, OUT_OF_RANGE, FUTURE_DATE, NOT_CONFIGURED, INVALID.
+ New reasons may be added; treat an unknown reason as INVALID.
+ x-known-values:
+ - REQUIRED
+ - REQUIRES_FIELD
+ - UNPARSEABLE
+ - OUT_OF_RANGE
+ - FUTURE_DATE
+ - NOT_CONFIGURED
+ - INVALID
limit:
type: number
Validate:
diff --git a/scripts/submissions/check_contract.py b/scripts/submissions/check_contract.py
index 48e8222ce7..68448094ef 100644
--- a/scripts/submissions/check_contract.py
+++ b/scripts/submissions/check_contract.py
@@ -88,6 +88,12 @@ def resolve(value):
("/api/v3/submissions/{id}/files", "post"),
("/api/v3/submissions/{id}/validate", "post")]:
assert "ETag" in expanded["paths"][path][method]["responses"]["200"]["headers"]
+import re
+java_reasons = re.search(r"REASONS = List\.of\(([^)]*)\)", (ROOT / "src/main/java/org/ecocean/api/submission/SubmissionValueIssues.java").read_text()).group(1)
+java_reasons = re.findall(r'"([A-Z_]+)"', java_reasons)
+published_issue = yaml.safe_load((ROOT / "src/main/resources/openapi.yaml").read_text())["components"]["schemas"]["SubmissionApiIssue"]["properties"]["reason"]
+assert published_issue == spec["components"]["schemas"]["Issue"]["properties"]["reason"], "issue reason schemas differ"
+assert published_issue["x-known-values"] == java_reasons, (published_issue["x-known-values"], java_reasons)
print(f"Checked {len(operations)} operations and all local references.")
diff --git a/src/main/java/org/ecocean/api/submission/SubmissionValidator.java b/src/main/java/org/ecocean/api/submission/SubmissionValidator.java
index fd1f075be3..dc5fd84e53 100644
--- a/src/main/java/org/ecocean/api/submission/SubmissionValidator.java
+++ b/src/main/java/org/ecocean/api/submission/SubmissionValidator.java
@@ -9,6 +9,9 @@
/** Strict new-encounter boundary around the existing bulk field validators. */
public class SubmissionValidator {
+ private final java.time.Clock clock;
+ public SubmissionValidator() { this(java.time.Clock.systemUTC()); }
+ SubmissionValidator(java.time.Clock clock) { this.clock = clock; } // tests fix "today"
public static final Set FIELDS = Set.of("Encounter.genus", "Encounter.specificEpithet",
"Encounter.year", "Encounter.month", "Encounter.day", "Encounter.hour", "Encounter.minutes",
"Encounter.locationID", "Encounter.decimalLatitude", "Encounter.decimalLongitude",
@@ -59,15 +62,20 @@ public JSONObject validate(Submission draft, Shepherd sh, SubmissionFiles storag
// No explicit encounter IDs are accepted: the importer creates one encounter per row.
if (media.isEmpty()) issue(errors, source, i, "Encounter.mediaAsset0", "REQUIRED_VALUE", "At least one image required");
if (media.size() > config.getInt("maxMediaPerEncounter")) issue(errors, source, i, null, "LIMIT_EXCEEDED", "Too many images for one encounter");
- if (!(fields.opt("Encounter.locationID") instanceof String) || !configuredLocation(config.getJSONObject("locations"), fields.optString("Encounter.locationID", null)))
- issue(errors, source, i, "Encounter.locationID", "INVALID_LOCATION", "A configured location ID is required");
+ boolean locationRejected = !(fields.opt("Encounter.locationID") instanceof String) || !configuredLocation(config.getJSONObject("locations"), fields.optString("Encounter.locationID", null));
+ if (locationRejected) issue(errors, source, i, "Encounter.locationID", "INVALID_LOCATION", "A configured location ID is required");
+ // captured before legacy validation so every date it judges future is also after this date
+ java.time.LocalDate today = clock.instant().atOffset(Util.LATEST_CIVIL_OFFSET).toLocalDate();
Map checked = BulkImportUtil.validateRow(copied, sh);
for (Map.Entry entry : checked.entrySet()) {
if (entry.getValue() instanceof BulkValidator) {
Object value = ((BulkValidator)entry.getValue()).getValue();
if (value != null) values.put(entry.getKey(), value);
- else issue(errors, source, i, entry.getKey(), "INVALID_VALUE", "Provided value cannot be empty or unparseable");
- } else issue(errors, source, i, entry.getKey(), "INVALID_VALUE", "Value failed bulk-import validation");
+ else issue(errors, source, i, entry.getKey(), "INVALID_VALUE", "INVALID", "Provided value cannot be empty");
+ } else if (!(locationRejected && "Encounter.locationID".equals(entry.getKey()))) { // already INVALID_LOCATION
+ SubmissionValueIssues.Issue explained = SubmissionValueIssues.explain(entry.getKey(), (Exception)entry.getValue(), copied, checked, sh, today);
+ issue(errors, source, i, explained.field, "INVALID_VALUE", explained.reason, explained.message);
+ }
}
normalized.put(new JSONObject().put("clientRowId", source).put("fields", values));
}
@@ -79,7 +87,11 @@ public JSONObject validate(Submission draft, Shepherd sh, SubmissionFiles storag
.put("effectiveOwnerId", draft.getOwnerId()).put("processing", new JSONObject().put("mode", draft.getProcessingMode()));
}
private static void issue(JSONArray issues, String source, int row, String field, String code, String message) {
+ issue(issues, source, row, field, code, null, message);
+ }
+ private static void issue(JSONArray issues, String source, int row, String field, String code, String reason, String message) {
JSONObject issue = new JSONObject().put("code", code).put("message", message);
+ if (reason != null) issue.put("reason", reason);
if (source != null) issue.put("clientRowId", source).put("rowIndex", row);
if (field != null) issue.put("field", field);
issues.put(issue);
diff --git a/src/main/java/org/ecocean/api/submission/SubmissionValueIssues.java b/src/main/java/org/ecocean/api/submission/SubmissionValueIssues.java
new file mode 100644
index 0000000000..536ae42d98
--- /dev/null
+++ b/src/main/java/org/ecocean/api/submission/SubmissionValueIssues.java
@@ -0,0 +1,140 @@
+package org.ecocean.api.submission;
+
+import java.time.LocalDate;
+import java.util.*;
+import org.ecocean.CommonConfiguration;
+import org.ecocean.Util;
+import org.ecocean.api.SiteSettings;
+import org.ecocean.api.bulk.BulkValidator;
+import org.ecocean.shepherd.core.Shepherd;
+import org.json.JSONObject;
+
+/**
+ * Explains a legacy bulk-import field rejection as a stable submissions reason, a specific message and the field
+ * the submitter should look at. Explanation never changes whether a row is accepted: callers report exactly one
+ * issue per legacy rejection. Mapping keys on legacy message text, pinned by SubmissionValueIssuesTest, with an
+ * INVALID fallback that never exposes Java exception text.
+ */
+public class SubmissionValueIssues {
+ public static final List REASONS = List.of("REQUIRED", "REQUIRES_FIELD", "UNPARSEABLE", "OUT_OF_RANGE",
+ "FUTURE_DATE", "NOT_CONFIGURED", "INVALID");
+ private static final int MAX_ECHO = 64, MAX_ALLOWED_LISTED = 20, MAX_ALLOWED_TEXT = 300;
+
+ public static class Issue {
+ public final String field, reason, message;
+ Issue(String field, String reason, String message) { this.field = field; this.reason = reason; this.message = message; }
+ }
+
+ /**
+ * @param today the date at UTC+14 captured before legacy validation ran; legacy "future" was judged at the
+ * same or a later instant, so every date it rejected is also after this date
+ */
+ public static Issue explain(String field, Exception ex, JSONObject fields, Map checked,
+ Shepherd sh, LocalDate today) {
+ String legacy = ex.getMessage() == null ? "" : ex.getMessage();
+ Object raw = fields.opt(field);
+ if (legacy.startsWith("required value") || legacy.equals("must supply a valid month along with day")) {
+ // legacy replaces a supplied-but-invalid value's own error with a generic "required"; recover it
+ if (raw != null) {
+ Issue cause = recover(field, raw, sh);
+ if (cause != null) return cause;
+ }
+ if (legacy.startsWith("must supply")) return new Issue(field, "REQUIRES_FIELD", "Encounter.month is required when Encounter.day is supplied");
+ return new Issue(field, "REQUIRED", field + " is required");
+ }
+ if (legacy.equals("date is in the future")) return futureDate(field, fields, checked, today);
+ return translate(field, legacy, fields, checked, sh);
+ }
+
+ private static Issue recover(String field, Object raw, Shepherd sh) {
+ try {
+ BulkValidator.validateValue(field, raw, sh);
+ return null;
+ } catch (Exception ex) {
+ String message = ex.getMessage() == null ? "" : ex.getMessage();
+ if (message.startsWith("required value")) return null;
+ return translate(field, message, new JSONObject().put(field, raw), Collections.emptyMap(), sh);
+ }
+ }
+
+ private static Issue translate(String field, String legacy, JSONObject fields, Map checked, Shepherd sh) {
+ if (legacy.startsWith("error parsing integer")) return new Issue(field, "UNPARSEABLE", field + " must be a whole number");
+ if (legacy.startsWith("error parsing double")) return new Issue(field, "UNPARSEABLE", field + " must be a decimal number");
+ if (legacy.equals("year value too small")) return new Issue(field, "OUT_OF_RANGE", "Encounter.year must be 1000 or later");
+ if (legacy.startsWith("month value too")) return new Issue(field, "OUT_OF_RANGE", "Encounter.month must be 1 through 12");
+ if (legacy.startsWith("day value too")) return new Issue(field, "OUT_OF_RANGE", "Encounter.day must be 1 through 31");
+ if (legacy.startsWith("hour value too")) return new Issue(field, "OUT_OF_RANGE", "Encounter.hour must be 0 through 23");
+ if (legacy.startsWith("minutes value too")) return new Issue(field, "OUT_OF_RANGE", "Encounter.minutes must be 0 through 59");
+ if (legacy.startsWith("invalid Encounter.decimalLatitude value")) return new Issue(field, "OUT_OF_RANGE", "Encounter.decimalLatitude must be between -90 and 90");
+ if (legacy.startsWith("invalid Encounter.decimalLongitude value")) return new Issue(field, "OUT_OF_RANGE", "Encounter.decimalLongitude must be between -180 and 180");
+ if (legacy.equals("day is out of range for month")) {
+ Integer y = legacyComponent(fields, checked, "Encounter.year"), m = legacyComponent(fields, checked, "Encounter.month");
+ String day = echo(fields.opt("Encounter.day"));
+ if (y != null && m != null) return new Issue(field, "OUT_OF_RANGE", String.format("%04d-%02d has no day %s", y, m, day));
+ return new Issue(field, "OUT_OF_RANGE", "Encounter.day does not exist in the supplied month");
+ }
+ if (legacy.equals("must supply both latitude and longitude"))
+ return new Issue(field, "REQUIRES_FIELD", "Encounter.decimalLatitude and Encounter.decimalLongitude must be supplied together");
+ if (legacy.equals("invalid taxonomy value")) {
+ String name = Util.taxonomyString(String.valueOf(fields.opt("Encounter.genus")), String.valueOf(fields.opt("Encounter.specificEpithet")));
+ return new Issue(field, "NOT_CONFIGURED", "'" + echo(name) + "' is not a configured taxonomy; see siteTaxonomies in /api/v3/site-settings");
+ }
+ if (legacy.startsWith("invalid location value"))
+ return new Issue(field, "NOT_CONFIGURED", "'" + echo(fields.opt(field)) + "' is not a configured location ID");
+ if (legacy.startsWith("invalid sex value")) return notConfigured(field, fields, Arrays.asList(SiteSettings.VALUES_SEX));
+ if (legacy.startsWith("invalid lifeStage value"))
+ return notConfigured(field, fields, CommonConfiguration.getIndexedPropertyValues("lifeStage", sh.getContext()));
+ if (legacy.startsWith("invalid livingStatus value"))
+ return notConfigured(field, fields, CommonConfiguration.getIndexedPropertyValues("livingStatus", sh.getContext()));
+ return new Issue(field, "INVALID", sanitize(legacy));
+ }
+
+ private static Issue futureDate(String field, JSONObject fields, Map checked, LocalDate today) {
+ // compare the components exactly as legacy checkYMD did, even where a later check replaced their entries
+ Integer year = legacyComponent(fields, checked, "Encounter.year"), month = legacyComponent(fields, checked, "Encounter.month"),
+ day = legacyComponent(fields, checked, "Encounter.day");
+ String suffix = " is later than today's date in every time zone";
+ if (year == null) return new Issue(field, "FUTURE_DATE", "The date" + suffix);
+ if (year > today.getYear()) return new Issue("Encounter.year", "FUTURE_DATE", String.format("%04d", year) + suffix);
+ if (month != null && year == today.getYear()) {
+ if (month > today.getMonthValue())
+ return new Issue("Encounter.month", "FUTURE_DATE", String.format("%04d-%02d", year, month) + suffix);
+ if (day != null && month == today.getMonthValue() && day > today.getDayOfMonth())
+ return new Issue("Encounter.day", "FUTURE_DATE", String.format("%04d-%02d-%02d", year, month, day) + suffix);
+ }
+ return new Issue(field, "FUTURE_DATE", "The date" + suffix); // defensive: legacy judged a later instant
+ }
+
+ // A component legacy checkYMD compared: its validated value, or, when a later date check replaced the entry
+ // (year by "date is in the future", day by "day is out of range for month"), the same raw value re-parsed.
+ private static Integer legacyComponent(JSONObject fields, Map checked, String field) {
+ Object entry = checked.get(field);
+ if (entry instanceof BulkValidator) {
+ Object value = ((BulkValidator)entry).getValue();
+ return value instanceof Integer ? (Integer)value : null;
+ }
+ String message = entry instanceof Exception ? ((Exception)entry).getMessage() : null;
+ if (!"date is in the future".equals(message) && !"day is out of range for month".equals(message)) return null;
+ try { return Integer.valueOf(String.valueOf(fields.opt(field))); }
+ catch (NumberFormatException ex) { return null; }
+ }
+
+ private static Issue notConfigured(String field, JSONObject fields, List allowed) {
+ String message = "'" + echo(fields.opt(field)) + "' is not a configured " + field.substring("Encounter.".length()) + " value";
+ String list = allowed == null || allowed.isEmpty() || allowed.size() > MAX_ALLOWED_LISTED ? null : String.join(", ", allowed);
+ if (list != null && list.length() <= MAX_ALLOWED_TEXT) message += "; use one of: " + list;
+ else message += "; see /api/v3/site-settings for configured values";
+ return new Issue(field, "NOT_CONFIGURED", message);
+ }
+
+ static String echo(Object value) {
+ String text = String.valueOf(value);
+ return text.length() <= MAX_ECHO ? text : text.substring(0, MAX_ECHO) + "...";
+ }
+
+ static String sanitize(String legacy) {
+ int java = legacy.indexOf("java.");
+ String text = (java >= 0 ? legacy.substring(0, java) : legacy).replaceAll("[\\s:]+$", "");
+ return text.isEmpty() ? "Value is invalid" : echo(text);
+ }
+}
diff --git a/src/main/resources/agent-skills/submit-sightings.md b/src/main/resources/agent-skills/submit-sightings.md
index b1aaef5009..024ac2ed72 100644
--- a/src/main/resources/agent-skills/submit-sightings.md
+++ b/src/main/resources/agent-skills/submit-sightings.md
@@ -290,7 +290,8 @@ An illustrative excerpt of a failed report is:
"rowIndex": 0,
"field": "Encounter.month",
"code": "INVALID_VALUE",
- "message": "Value failed bulk-import validation"
+ "reason": "OUT_OF_RANGE",
+ "message": "Encounter.month must be 1 through 12"
}
]
}
@@ -301,20 +302,40 @@ The actual report also has IDs, digests, normalized rows and processing metadata
file-level errors omit row and field. One bad value may produce several issues;
do not depend on issue order or expect exactly one error per field.
+Read `code` first, then the optional `reason` on `INVALID_VALUE` issues. It says
+why the value was rejected:
+
+| `reason` | Meaning |
+|---|---|
+| `REQUIRED` | A required field is missing. |
+| `REQUIRES_FIELD` | This field is needed because another was supplied: a month for a day, or the other coordinate. |
+| `UNPARSEABLE` | Not a number of the expected kind, for example text or a fraction in a whole-number field. |
+| `OUT_OF_RANGE` | A number outside the allowed range, or a day that does not exist in that month. |
+| `FUTURE_DATE` | The date is later than today everywhere on Earth. The issue names the part (year, month or day) that is too late. |
+| `NOT_CONFIGURED` | Not one of this installation's configured values. The message names the value and, for short lists, the allowed values. |
+| `INVALID` | Another rejection; read the message. |
+
+New reasons may be added later; treat an unrecognized or absent `reason` like
+`INVALID`. The issue shows where a problem was detected, not proof of which source
+value is wrong: check the original observation before changing anything, and never
+change a correct value just to pass validation.
+
Concrete failure examples and corrections (assume other fields are valid):
| Input problem | Expected validation issue | Correction |
|---|---|---|
| locationID is `"Reef near town"`, but that is not a configured ID; or locationID is missing | `INVALID_LOCATION` on `Encounter.locationID` | Obtain the actual corresponding ID; do not substitute an unrelated location. |
-| genus/epithet pair is not in configured taxonomies, or either required field is absent | `INVALID_VALUE` on the affected taxonomy field(s) | Use the correct configured scientific components or ask the operator to address missing configuration. |
-| `Encounter.month: 13` | `INVALID_VALUE` on `Encounter.month` | Correct from source evidence; omit only if genuinely unknown. |
-| year 2025, month 2, day 29 | `INVALID_VALUE` on `Encounter.day` | 2025 is not a leap year; correct the date from the original observation. |
-| day 18 with no month | `INVALID_VALUE` on `Encounter.month` | Supply the known month, or preserve only the date precision actually known. |
-| year 999, a nonnumeric year, missing year, or a future observation date | `INVALID_VALUE` on `Encounter.year` | Provide a real past/current observation year/date. |
-| hour 24 or minutes 60 | `INVALID_VALUE` on that field | Use 24-hour components in range; do not guess missing time. |
-| latitude 91, or latitude supplied without longitude | `INVALID_VALUE` on latitude or the missing longitude | Provide both valid decimal-degree coordinates, or omit both if unknown. |
-| sex `"F"`, `"Female"`, or `"M"` | `INVALID_VALUE` on `Encounter.sex` | Use exact `female`, `male`, or `unknown` when supported by source evidence. |
-| lifeStage `"juvenile"` when absent from configured lifeStage values; likewise an unconfigured livingStatus | `INVALID_VALUE` on the corresponding field | Map only to a semantically correct configured value; otherwise ask or omit an unknown optional value. |
+| genus/epithet pair is not in configured taxonomies | `INVALID_VALUE` (`NOT_CONFIGURED`) on both taxonomy fields | Use the correct configured scientific components or ask the operator to address missing configuration. |
+| genus or specificEpithet absent | `INVALID_VALUE` (`REQUIRED`) on the missing field | Supply the configured scientific component from the source record. |
+| `Encounter.month: 13` | `INVALID_VALUE` (`OUT_OF_RANGE`) on `Encounter.month` | Correct from source evidence; omit only if genuinely unknown. |
+| year 2025, month 2, day 29 | `INVALID_VALUE` (`OUT_OF_RANGE`) on `Encounter.day` | 2025 is not a leap year; correct the date from the original observation. |
+| day 18 with no month | `INVALID_VALUE` (`REQUIRES_FIELD`) on `Encounter.month` | Supply the known month, or preserve only the date precision actually known. |
+| year 999; a nonnumeric year; missing year | `INVALID_VALUE` (`OUT_OF_RANGE`, `UNPARSEABLE` or `REQUIRED`) on `Encounter.year` | Provide the real observation year. |
+| a future observation date | `INVALID_VALUE` (`FUTURE_DATE`) on the year, month or day that is too late | Check that part of the date against the original observation. |
+| hour 24 or minutes 60 | `INVALID_VALUE` (`OUT_OF_RANGE`) on that field | Use 24-hour components in range; do not guess missing time. |
+| latitude 91, or latitude supplied without longitude | `INVALID_VALUE` (`OUT_OF_RANGE`) on latitude, or (`REQUIRES_FIELD`) on the missing longitude | Provide both valid decimal-degree coordinates, or omit both if unknown. |
+| sex `"F"`, `"Female"`, or `"M"` | `INVALID_VALUE` (`NOT_CONFIGURED`) on `Encounter.sex` | Use exact `female`, `male`, or `unknown` when supported by source evidence. |
+| lifeStage `"juvenile"` when absent from configured lifeStage values; likewise an unconfigured livingStatus | `INVALID_VALUE` (`NOT_CONFIGURED`) on the corresponding field | Map only to a semantically correct configured value; otherwise ask or omit an unknown optional value. |
| mediaAsset0 `"Photo.JPG"` when the completed upload is `"photo.jpg"` | `MISSING_MEDIA` (and possibly `REQUIRED_VALUE`) | Match the exact manifest filename and ensure its upload completed. |
| no image reference | `REQUIRED_VALUE` on `Encounter.mediaAsset0` | Upload and reference at least one authorized photo. |
| same image in two slots or rows | `DUPLICATE_MEDIA` | Put each uploaded image in exactly one slot in one row. |
diff --git a/src/main/resources/openapi.yaml b/src/main/resources/openapi.yaml
index f16d94f0ae..0df5a1df14 100644
--- a/src/main/resources/openapi.yaml
+++ b/src/main/resources/openapi.yaml
@@ -504,6 +504,21 @@ components:
field:
type: string
minLength: 1
+ reason:
+ type: string
+ minLength: 1
+ description: >-
+ Present on INVALID_VALUE issues: why the value was rejected. One of REQUIRED,
+ REQUIRES_FIELD, UNPARSEABLE, OUT_OF_RANGE, FUTURE_DATE, NOT_CONFIGURED, INVALID.
+ New reasons may be added; treat an unknown reason as INVALID.
+ x-known-values:
+ - REQUIRED
+ - REQUIRES_FIELD
+ - UNPARSEABLE
+ - OUT_OF_RANGE
+ - FUTURE_DATE
+ - NOT_CONFIGURED
+ - INVALID
limit:
type: number
SubmissionApiCapabilities:
diff --git a/src/test/java/org/ecocean/api/submission/SubmissionValueIssuesTest.java b/src/test/java/org/ecocean/api/submission/SubmissionValueIssuesTest.java
new file mode 100644
index 0000000000..978433cc8a
--- /dev/null
+++ b/src/test/java/org/ecocean/api/submission/SubmissionValueIssuesTest.java
@@ -0,0 +1,185 @@
+package org.ecocean.api.submission;
+
+import java.nio.file.Path;
+import java.time.LocalDate;
+import java.util.*;
+import org.ecocean.*;
+import org.ecocean.shepherd.core.Shepherd;
+import org.ecocean.submission.Submission;
+import org.json.*;
+import org.junit.jupiter.api.*;
+import org.junit.jupiter.api.io.TempDir;
+import org.mockito.MockedStatic;
+import static org.mockito.Mockito.*;
+import static org.junit.jupiter.api.Assertions.*;
+
+/** Runs the real legacy field validators, so a legacy wording change fails here instead of degrading reasons. */
+class SubmissionValueIssuesTest {
+ @TempDir Path root;
+
+ private static final class Case {
+ final String id, field, reason, messagePart; final JSONObject changes; final String[] drop;
+ Case(String id, JSONObject changes, String field, String reason, String messagePart, String... drop) {
+ this.id = id; this.changes = changes; this.field = field; this.reason = reason; this.messagePart = messagePart; this.drop = drop;
+ }
+ }
+
+ private static JSONObject set(Object... pairs) {
+ JSONObject value = new JSONObject();
+ for (int i = 0; i < pairs.length; i += 2) value.put((String)pairs[i], pairs[i + 1]);
+ return value;
+ }
+
+ private static String ymd(LocalDate date) {
+ return String.format("%04d-%02d-%02d", date.getYear(), date.getMonthValue(), date.getDayOfMonth());
+ }
+
+ @Test void legacyRejectionsReportSpecificReasonMessageAndField() throws Exception {
+ List cases = List.of(
+ new Case("year-missing", set(), "Encounter.year", "REQUIRED", "Encounter.year is required", "Encounter.year", "Encounter.month", "Encounter.day"),
+ new Case("genus-missing", set(), "Encounter.genus", "REQUIRED", "Encounter.genus is required", "Encounter.genus"),
+ new Case("year-999", set("Encounter.year", 999), "Encounter.year", "OUT_OF_RANGE", "1000 or later"),
+ new Case("year-text", set("Encounter.year", "abc"), "Encounter.year", "UNPARSEABLE", "whole number"),
+ new Case("year-fraction", set("Encounter.year", 2017.5), "Encounter.year", "UNPARSEABLE", "whole number"),
+ new Case("month-13", set("Encounter.month", 13), "Encounter.month", "OUT_OF_RANGE", "1 through 12"),
+ new Case("day-32", set("Encounter.day", 32), "Encounter.day", "OUT_OF_RANGE", "1 through 31"),
+ new Case("feb-29-2025", set("Encounter.year", 2025, "Encounter.month", 2, "Encounter.day", 29), "Encounter.day", "OUT_OF_RANGE", "2025-02 has no day 29"),
+ new Case("day-without-month", set(), "Encounter.month", "REQUIRES_FIELD", "required when Encounter.day", "Encounter.month"),
+ new Case("hour-24", set("Encounter.hour", 24), "Encounter.hour", "OUT_OF_RANGE", "0 through 23"),
+ new Case("minutes-60", set("Encounter.hour", 10, "Encounter.minutes", 60), "Encounter.minutes", "OUT_OF_RANGE", "0 through 59"),
+ new Case("latitude-91", set("Encounter.decimalLatitude", 91, "Encounter.decimalLongitude", 36.9), "Encounter.decimalLatitude", "OUT_OF_RANGE", "-90 and 90"),
+ new Case("latitude-nan", set("Encounter.decimalLatitude", "NaN", "Encounter.decimalLongitude", 36.9), "Encounter.decimalLatitude", "OUT_OF_RANGE", "-90 and 90"),
+ new Case("longitude-181", set("Encounter.decimalLatitude", 0.29, "Encounter.decimalLongitude", 181), "Encounter.decimalLongitude", "OUT_OF_RANGE", "-180 and 180"),
+ new Case("latitude-text", set("Encounter.decimalLatitude", "abc", "Encounter.decimalLongitude", 36.9), "Encounter.decimalLatitude", "UNPARSEABLE", "decimal number"),
+ new Case("latitude-alone", set("Encounter.decimalLatitude", 0.29), "Encounter.decimalLongitude", "REQUIRES_FIELD", "supplied together"),
+ new Case("taxonomy", set("Encounter.specificEpithet", "quagga"), "Encounter.genus", "NOT_CONFIGURED", "'Equus quagga' is not a configured taxonomy"),
+ new Case("sex", set("Encounter.sex", "F"), "Encounter.sex", "NOT_CONFIGURED", "'F' is not a configured sex value; use one of: unknown, male, female"),
+ new Case("life-stage", set("Encounter.lifeStage", "juvenile"), "Encounter.lifeStage", "NOT_CONFIGURED", "'juvenile' is not a configured lifeStage value"),
+ new Case("living-status", set("Encounter.livingStatus", "zombie"), "Encounter.livingStatus", "NOT_CONFIGURED", "'zombie' is not a configured livingStatus value"),
+ new Case("future-year", set("Encounter.year", 2027, "Encounter.month", 1, "Encounter.day", 1), "Encounter.year", "FUTURE_DATE", "2027 is later than today's date in every time zone"),
+ new Case("future-year-string", set("Encounter.year", "2027", "Encounter.month", 1, "Encounter.day", 1), "Encounter.year", "FUTURE_DATE", "2027 is later"),
+ new Case("future-year-only", set("Encounter.year", 2027), "Encounter.year", "FUTURE_DATE", "2027 is later", "Encounter.month", "Encounter.day"),
+ new Case("future-year-bad-month", set("Encounter.year", 2027, "Encounter.month", 13), "Encounter.year", "FUTURE_DATE", "2027 is later", "Encounter.day"),
+ new Case("future-month", set("Encounter.year", 2026, "Encounter.month", 10), "Encounter.month", "FUTURE_DATE", "2026-10 is later", "Encounter.day"),
+ new Case("future-day", set("Encounter.year", 2026, "Encounter.month", 9, "Encounter.day", 26), "Encounter.day", "FUTURE_DATE", "2026-09-26 is later"),
+ new Case("future-nonexistent-day", set("Encounter.year", 2026, "Encounter.month", 9, "Encounter.day", 31), "Encounter.day", "FUTURE_DATE", "2026-09-31 is later"),
+ new Case("past-day-this-month", set("Encounter.year", 2026, "Encounter.month", 9, "Encounter.day", 24), null, null, null),
+ new Case("legacy-only-location", set("Encounter.locationID", "retired"), "Encounter.locationID", "NOT_CONFIGURED", "'retired' is not a configured location ID"),
+ new Case("today", set("Encounter.year", 2026, "Encounter.month", 9, "Encounter.day", 25), null, null, null));
+ JSONArray errors = validate(cases).getJSONArray("errors");
+ String all = errors.toString();
+ assertFalse(all.contains("java."), all);
+ assertFalse(all.contains("Value failed bulk-import validation"), all);
+ for (Case c : cases) {
+ List row = new ArrayList<>();
+ for (int i = 0; i < errors.length(); i++) if (c.id.equals(errors.getJSONObject(i).optString("clientRowId"))) row.add(errors.getJSONObject(i));
+ if (c.field == null) { assertTrue(row.isEmpty(), c.id + " " + row); continue; }
+ JSONObject match = null;
+ for (JSONObject issue : row) if (c.field.equals(issue.optString("field")) && c.reason.equals(issue.optString("reason"))) match = issue;
+ assertNotNull(match, c.id + " expected " + c.field + "/" + c.reason + " in " + row);
+ assertEquals("INVALID_VALUE", match.getString("code"), c.id);
+ assertTrue(match.getString("message").contains(c.messagePart), c.id + ": " + match);
+ if ("FUTURE_DATE".equals(c.reason)) // exactly one future-date issue, on the component that is future
+ assertEquals(1, row.stream().filter(e -> "FUTURE_DATE".equals(e.optString("reason"))).count(), c.id + " " + row);
+ }
+ List badMonth = new ArrayList<>();
+ for (int i = 0; i < errors.length(); i++) if ("future-year-bad-month".equals(errors.getJSONObject(i).optString("clientRowId"))) badMonth.add(errors.getJSONObject(i));
+ assertTrue(badMonth.toString().contains("\"reason\":\"OUT_OF_RANGE\""), "invalid month still reported alongside future year: " + badMonth);
+ long nonexistentDay = issuesFor(errors, "future-nonexistent-day").stream()
+ .filter(e -> "Encounter.day".equals(e.optString("field")) && "OUT_OF_RANGE".equals(e.optString("reason"))
+ && e.getString("message").equals("2026-09 has no day 31")).count();
+ assertEquals(1, nonexistentDay, "nonexistent day still reported alongside future date");
+ }
+
+ @Test void unconfiguredLocationIsReportedOnce() throws Exception {
+ JSONArray errors = validate(List.of(new Case("location", set("Encounter.locationID", "Mpala"), null, null, null))).getJSONArray("errors");
+ int location = 0;
+ for (int i = 0; i < errors.length(); i++) if ("Encounter.locationID".equals(errors.getJSONObject(i).optString("field"))) location++;
+ assertEquals(1, location, errors.toString());
+ assertEquals("INVALID_LOCATION", errors.getJSONObject(0).getString("code"));
+ }
+
+ private static List issuesFor(JSONArray errors, String id) {
+ List row = new ArrayList<>();
+ for (int i = 0; i < errors.length(); i++) if (id.equals(errors.getJSONObject(i).optString("clientRowId"))) row.add(errors.getJSONObject(i));
+ return row;
+ }
+
+ @Test void longAllowedValueListsAreNotEchoed() throws Exception {
+ List longValues = new ArrayList<>();
+ for (int i = 0; i < 10; i++) longValues.add("stage-" + i + "-" + "x".repeat(60));
+ JSONArray errors = validate(List.of(new Case("life-stage", set("Encounter.lifeStage", "juvenile"), null, null, null)), longValues)
+ .getJSONArray("errors");
+ String message = errors.getJSONObject(0).getString("message");
+ assertTrue(message.endsWith("see /api/v3/site-settings for configured values"), message);
+ assertTrue(message.length() < 200, message);
+ }
+
+ @Test void unmappedMessagesFallBackWithoutJavaText() {
+ assertEquals("error parsing long", SubmissionValueIssues.sanitize("error parsing long: java.lang.NumberFormatException: For input string: \"x\""));
+ assertEquals("Value is invalid", SubmissionValueIssues.sanitize("java.lang.IllegalStateException"));
+ SubmissionValueIssues.Issue issue = SubmissionValueIssues.explain("Encounter.behavior", new IllegalStateException("something new"),
+ new JSONObject(), Collections.emptyMap(), mock(Shepherd.class), LocalDate.now());
+ assertEquals("INVALID", issue.reason);
+ assertEquals("something new", issue.message);
+ assertEquals("Encounter.behavior", issue.field);
+ }
+
+ private static final LocalDate TODAY = LocalDate.of(2026, 9, 25);
+ private static final java.time.Clock CLOCK = java.time.Clock.fixed(java.time.Instant.parse("2026-09-24T12:00:00Z"), java.time.ZoneOffset.UTC);
+
+ private JSONObject validate(List cases) throws Exception { return validate(cases, List.of()); }
+
+ /** Validates one draft of cases with "today" fixed for both the report and legacy validation, and checks
+ * the report invariants: exactly one issue per legacy rejection (except the duplicate location one), and
+ * normalized rows holding exactly the values legacy validation accepted. */
+ private JSONObject validate(List cases, List lifeStages) throws Exception {
+ assertEquals(TODAY, CLOCK.instant().atOffset(Util.LATEST_CIVIL_OFFSET).toLocalDate());
+ try (MockedStatic config = mockStatic(CommonConfiguration.class);
+ MockedStatic location = mockStatic(LocationID.class);
+ MockedStatic util = mockStatic(Util.class, CALLS_REAL_METHODS)) {
+ util.when(() -> Util.dateIsInFuture(any(), any(), any())).thenAnswer(a -> Util.dateIsInFuture(
+ a.getArgument(0), a.getArgument(1), a.getArgument(2), TODAY));
+ config.when(() -> CommonConfiguration.getMaxMediaCountEncounter(any())).thenReturn(10);
+ config.when(() -> CommonConfiguration.getIndexedPropertyValues(eq("lifeStage"), nullable(String.class))).thenReturn(lifeStages);
+ // "retired" is in the submissions location tree but rejected by the legacy location check
+ location.when(LocationID::getLocationIDStructure).thenReturn(new JSONObject("{\"locationID\":[{\"id\":\"reef\"},{\"id\":\"retired\"}]}"));
+ location.when(() -> LocationID.isValidLocationID("reef")).thenReturn(true);
+ Shepherd sh = mock(Shepherd.class);
+ when(sh.isValidTaxonomyName(anyString())).thenAnswer(a -> "Equus grevyi".equals(a.getArgument(0)));
+ SubmissionFiles storage = new SubmissionFiles(root);
+ JSONArray files = new JSONArray(), rows = new JSONArray();
+ for (int i = 0; i < cases.size(); i++) {
+ Case c = cases.get(i);
+ files.put(storage.write("img" + i + ".png", new java.io.ByteArrayInputStream(SubmissionFilesTest.png()), 10000));
+ JSONObject fields = set("Encounter.genus", "Equus", "Encounter.specificEpithet", "grevyi", "Encounter.year", 2017,
+ "Encounter.month", 4, "Encounter.day", 25, "Encounter.locationID", "reef", "Encounter.mediaAsset0", "img" + i + ".png");
+ for (String field : c.drop) fields.remove(field);
+ for (String key : c.changes.keySet()) fields.put(key, c.changes.get(key));
+ rows.put(new JSONObject().put("clientRowId", c.id).put("fields", fields));
+ }
+ Submission draft = new Submission("id", "context0", "owner", "hash", "hash", "{}", 0, Long.MAX_VALUE);
+ draft.setFiles(files.toString());
+ draft.replaceRows(rows.toString());
+ JSONObject report = new SubmissionValidator(CLOCK).validate(draft, sh, storage);
+ JSONArray errors = report.getJSONArray("errors");
+ assertEquals(errors.length() == 0, report.getBoolean("valid"));
+ for (int i = 0; i < rows.length(); i++) {
+ JSONObject row = rows.getJSONObject(i), fields = row.getJSONObject("fields");
+ Map legacy = org.ecocean.api.bulk.BulkImportUtil.validateRow(new JSONObject(fields.toString()), sh);
+ JSONObject accepted = new JSONObject();
+ int rejected = 0;
+ for (Map.Entry entry : legacy.entrySet()) {
+ if (entry.getValue() instanceof Exception) {
+ if (!("Encounter.locationID".equals(entry.getKey()) && "Mpala".equals(fields.opt("Encounter.locationID")))) rejected++;
+ } else accepted.put(entry.getKey(), ((org.ecocean.api.bulk.BulkValidator)entry.getValue()).getValue());
+ }
+ long reported = issuesFor(errors, row.getString("clientRowId")).stream().filter(e -> "INVALID_VALUE".equals(e.getString("code"))).count();
+ assertEquals(rejected, reported, row.getString("clientRowId") + " " + errors);
+ assertEquals(SubmissionJson.canonical(accepted),
+ SubmissionJson.canonical(report.getJSONArray("normalizedRows").getJSONObject(i).getJSONObject("fields")), row.getString("clientRowId"));
+ }
+ return report;
+ }
+ }
+}