diff --git a/.github/workflows/sbom-diff-and-risk-ci.yml b/.github/workflows/sbom-diff-and-risk-ci.yml index 118c2e2..941fdf1 100644 --- a/.github/workflows/sbom-diff-and-risk-ci.yml +++ b/.github/workflows/sbom-diff-and-risk-ci.yml @@ -24,10 +24,10 @@ jobs: working-directory: tools/sbom-diff-and-risk steps: - name: Check out repository - uses: actions/checkout@v6 + uses: actions/checkout@v4 - name: Set up Python - uses: actions/setup-python@v6 + uses: actions/setup-python@v5 with: python-version: "3.11" @@ -57,7 +57,7 @@ jobs: build-and-attest: # Keep provenance publication on trusted non-PR runs so consumers verify - # workflow-produced wheel and sdist artifacts from this repository workflow. + # workflow-produced wheel/sdist artifacts from this repository workflow. if: github.event_name != 'pull_request' needs: test runs-on: ubuntu-latest diff --git a/.github/workflows/sbom-diff-and-risk-code-scanning.yml b/.github/workflows/sbom-diff-and-risk-code-scanning.yml index d10067e..5f8ff65 100644 --- a/.github/workflows/sbom-diff-and-risk-code-scanning.yml +++ b/.github/workflows/sbom-diff-and-risk-code-scanning.yml @@ -18,7 +18,7 @@ jobs: working-directory: tools/sbom-diff-and-risk steps: - name: Check out repository - uses: actions/checkout@v6 + uses: actions/checkout@v5 - name: Set up Python uses: actions/setup-python@v6 diff --git a/tools/sbom-diff-and-risk/README.md b/tools/sbom-diff-and-risk/README.md index 85dd36c..c0a3ccc 100644 --- a/tools/sbom-diff-and-risk/README.md +++ b/tools/sbom-diff-and-risk/README.md @@ -1,6 +1,6 @@ # sbom-diff-and-risk -v0.2.0 adds policy-based enforcement, SARIF export, GitHub code scanning integration, and deterministic parser hardening for Python dependency inputs. +v0.3.0 adds opt-in PyPI provenance enrichment, provenance-aware policy and reporting, optional advisory Scorecard signals, and self-provenance verification guidance for workflow-built artifacts. `sbom-diff-and-risk` is a local, deterministic CLI for comparing two SBOMs or dependency manifests and producing JSON plus Markdown reports. @@ -156,9 +156,82 @@ sbom-diff-risk compare \ - `--warn-on rule[,rule...]` - `--strict` - `--enrich-pypi` +- `--pypi-timeout seconds` +- `--enrich-scorecard` +- `--scorecard-timeout seconds` - `--source-allowlist pypi.org,files.pythonhosted.org,github.com` -`--enrich-pypi` is reserved for future work and currently returns a clear error. +Offline mode remains the default. No network access occurs unless `--enrich-pypi` or `--enrich-scorecard` is set explicitly. + +## Opt-in Provenance Enrichment + +PyPI provenance and integrity enrichment is explicit and additive in this PR: + +- only Python / PyPI packages are queried +- no hidden network access occurs in default mode +- enrichment results are captured as evidence and summarized in the reports +- per-component `evidence.provenance` records stable lookup fields such as `supported`, `lookup_performed`, and per-file attestation totals +- lack of attestation is treated as unavailable metadata, not as proof of compromise +- policy evaluation can use these signals explicitly when configured +- SARIF stays conservative and only emits selected high-signal provenance policy violations + +When enabled, the tool queries PyPI-facing release metadata plus file-level provenance data and records stable evidence fields under component `evidence.provenance`, along with run metadata under `metadata.enrichment` and the top-level trust-signal report fields in the JSON report. + +```bash +sbom-diff-risk compare \ + --before examples/requirements_before.txt \ + --after examples/requirements_after.txt \ + --enrich-pypi \ + --pypi-timeout 3 \ + --out-json outputs/report-enriched.json +``` + +## Provenance-Aware Reporting + +When provenance enrichment is enabled, the reports surface trust signals directly instead of burying them in component evidence: + +- JSON includes `provenance_summary`, `attestation_summary`, `enrichment_metadata`, `trust_signal_notes`, and `provenance_policy_impact` +- Markdown includes `Provenance summary`, `Attestation gaps`, `Policy impact for provenance-related rules`, and `Trust signal notes` +- core diff semantics do not change when enrichment is enabled +- SARIF maps only selected high-signal provenance decisions such as `provenance_required`, blocking `missing_attestation`, and blocking `unverified_provenance` +- provenance-related SARIF alerts prefer file-level locations that point to the relevant compared manifest or SBOM input + +Routine enrichment outcomes remain JSON and Markdown evidence for review. Non-blocking enrichment facts do not automatically become SARIF alerts. + +## Opt-in Scorecard Enrichment + +OpenSSF Scorecard enrichment is also explicit and advisory: + +- no Scorecard requests are made unless `--enrich-scorecard` is set +- lookups only occur when a component can be mapped to a repository with high confidence from explicit metadata +- repository registry pages and ambiguous URLs are treated as unmapped instead of inferred +- Scorecard results are auxiliary trust signals, not proof of safety +- Scorecard-only SARIF alerts are emitted only when policy explicitly turns a threshold breach into a violation + +```bash +sbom-diff-risk compare \ + --before examples/cdx_before.json \ + --after examples/cdx_after.json \ + --enrich-scorecard \ + --scorecard-timeout 3 \ + --out-json outputs/report-scorecard.json +``` + +If you want policy gating, make it explicit with a v3 policy such as [policy-scorecard-minimal.yml](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/policy-scorecard-minimal.yml), which sets `minimum_scorecard_score` and opts into the `scorecard_below_threshold` rule. + +Setting `minimum_scorecard_score` alone is advisory metadata for review. It only affects policy outcomes when `scorecard_below_threshold` is configured explicitly in `block_on`, `warn_on`, or `ignore_rules`. + +## Self-provenance + +This repository also records provenance for `sbom-diff-and-risk` itself by generating GitHub artifact attestations for the wheel and source distribution produced by the `sbom-diff-and-risk-ci` workflow. + +- the attested files are the wheel and source distribution built by `python -m build` from `tools/sbom-diff-and-risk` +- the build files are uploaded together as the `sbom-diff-and-risk-dist` workflow artifact +- only trusted non-PR runs publish the attestation +- consumers can verify provenance with GitHub's attestation tooling after downloading one of those artifacts +- this complements the tool's analysis of third-party supply-chain inputs, but it does not replace that analysis + +See [docs/self-provenance.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/docs/self-provenance.md) for the exact attested filenames, where the evidence appears in GitHub, and a run-by-run verification flow for consumers. ## Examples @@ -167,11 +240,15 @@ The [examples/](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-an - before/after inputs for CycloneDX JSON, SPDX JSON, `requirements.txt`, and `pyproject.toml` - dependency-group examples at `examples/pyproject_groups_before.toml` and `examples/pyproject_groups_after.toml` - example policies at `examples/policy-minimal.yml` and `examples/policy-strict.yml` +- provenance-aware policy examples at `examples/policy-provenance-minimal.yml` and `examples/policy-provenance-strict.yml` +- a Scorecard-aware policy example at `examples/policy-scorecard-minimal.yml` - a sample pass JSON report at [sample-report.json](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-report.json) - a sample pass Markdown report at [sample-report.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-report.md) - sample policy-warn reports at [sample-policy-warn-report.json](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.json) and [sample-policy-warn-report.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.md) - sample policy-fail reports at [sample-policy-fail-report.json](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.json) and [sample-policy-fail-report.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.md) - a sample SARIF export at [sample-sarif.sarif](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-sarif.sarif) +- provenance-aware sample reports at [sample-provenance-report.json](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-provenance-report.json), [sample-provenance-report.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-provenance-report.md), and [sample-provenance-report.sarif](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-provenance-report.sarif) +- Scorecard-aware sample reports at [sample-scorecard-report.json](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-scorecard-report.json), [sample-scorecard-report.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-scorecard-report.md), and [sample-scorecard-report.sarif](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-scorecard-report.sarif) - requirements-based sample reports at [sample-requirements-report.json](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-requirements-report.json) and [sample-requirements-report.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-requirements-report.md) ## Enforcement Mode @@ -214,9 +291,10 @@ SARIF export is intentionally conservative. The current renderer emits a GitHub- - `suspicious_source` - `unknown_license` - `major_upgrade` -- selected blocking policy results such as `max_added_packages` and `allow_sources` +- selected policy results such as `max_added_packages`, `allow_sources`, `provenance_required`, and blocking provenance violations like `missing_attestation` or `unverified_provenance` +- explicit Scorecard policy violations such as `scorecard_below_threshold` -It does not turn every diff or informational heuristic into a code scanning alert. +It does not turn every enrichment fact, diff, or informational heuristic into a code scanning alert. ```bash sbom-diff-risk compare \ @@ -228,17 +306,8 @@ sbom-diff-risk compare \ For GitHub code scanning integration guidance and a minimal upload workflow, see [docs/github-code-scanning.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/docs/github-code-scanning.md). -## Self-provenance - -This repository also records provenance for `sbom-diff-and-risk` itself by generating GitHub artifact attestations for the wheel and source distribution produced by the `sbom-diff-and-risk-ci` workflow. +For details on how this repository attests the tool's own wheel and source distribution artifacts, see [docs/self-provenance.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/docs/self-provenance.md). -- the attested files are the wheel and source distribution built by `python -m build` from `tools/sbom-diff-and-risk` -- the build files are uploaded together as the `sbom-diff-and-risk-dist` workflow artifact -- only trusted non-PR runs publish the attestation -- consumers can verify provenance with GitHub's attestation tooling after downloading one of those artifacts -- this complements the tool's analysis of third-party supply-chain inputs, but it does not replace that analysis - -See [docs/self-provenance.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/docs/self-provenance.md) for the exact attested filenames, where the evidence appears in GitHub, and a run-by-run verification flow for consumers. ## Parser Boundaries Deterministic local mode intentionally supports a conservative subset of packaging syntax. The detailed matrix lives in [docs/parser-boundaries.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/docs/parser-boundaries.md). @@ -266,9 +335,12 @@ Deterministic local mode intentionally supports a conservative subset of packagi ## Limitations - default mode is local-file based only. +- PyPI provenance enrichment is opt-in only via `--enrich-pypi`; default runs stay offline. - `generated_at` remains `null` to preserve deterministic report output. - `stale_package` is not resolved offline. The report emits `not_evaluated` instead. -- SARIF export intentionally covers only a conservative subset of findings in v0.2. +- provenance evidence is recorded for supported PyPI packages only; unsupported and failed lookups remain explicit evidence gaps. +- SARIF export intentionally covers only a conservative subset of findings in v0.2, including only selected high-signal provenance policy violations. +- Scorecard enrichment is opt-in only via `--enrich-scorecard`, uses only high-confidence repository mappings, and remains advisory unless policy explicitly gates it. - No vulnerability database integration, CVE matching, or advisory enrichment. - `requirements.txt` support intentionally covers a conservative subset: plain PEP 508 requirement entries, comments, extras, markers, and line continuations. - `requirements.txt` intentionally rejects include/constraint directives, editable installs, direct URL/path refs, index/source options, and other pip-only install flags in deterministic mode. diff --git a/tools/sbom-diff-and-risk/docs/dependency-risk-heuristics.md b/tools/sbom-diff-and-risk/docs/dependency-risk-heuristics.md index 7a9b55e..3905beb 100644 --- a/tools/sbom-diff-and-risk/docs/dependency-risk-heuristics.md +++ b/tools/sbom-diff-and-risk/docs/dependency-risk-heuristics.md @@ -25,6 +25,7 @@ The current rules are intentionally conservative: ## Deferred work - real `stale_package` evaluation behind explicit enrichment +- provenance-based policy gates over opt-in enrichment evidence - ecosystem-specific trust rules - advisory and CVE enrichment - configurable risk policy profiles diff --git a/tools/sbom-diff-and-risk/docs/policy-schema.md b/tools/sbom-diff-and-risk/docs/policy-schema.md index 3c92457..2208e6e 100644 --- a/tools/sbom-diff-and-risk/docs/policy-schema.md +++ b/tools/sbom-diff-and-risk/docs/policy-schema.md @@ -1,15 +1,17 @@ # Policy schema -`sbom-diff-and-risk` supports a YAML-only policy schema in v1. +`sbom-diff-and-risk` supports YAML-only policy schemas in versions `1`, `2`, and `3` for the local, provenance-aware, and optional Scorecard-aware policy flows described here. The schema is intentionally conservative and fail-closed: - unknown rule ids are rejected - unknown top-level keys are rejected - invalid types are rejected -- only schema version `1` is supported +- version `1` remains the v0.2-compatible schema and existing v0.2 policies continue to work unchanged +- version `2` adds provenance-aware gating for explicit PyPI enrichment evidence +- version `3` adds optional Scorecard-aware gating for explicitly requested Scorecard enrichment -## Fields +## Version 1 fields - `version: 1` - `block_on: [rule_id, ...]` @@ -18,7 +20,7 @@ The schema is intentionally conservative and fail-closed: - `allow_sources: [host, ...]` - `ignore_rules: [rule_id, ...]` -## Supported rule ids +## Version 1 supported rule ids - `new_package` - `major_upgrade` @@ -29,6 +31,41 @@ The schema is intentionally conservative and fail-closed: - `max_added_packages` - `allow_sources` +## Version 2 fields + +Version `2` supports every version `1` field plus: + +- `require_attestations_for_new_packages: bool` +- `require_provenance_for_suspicious_sources: bool` +- `allow_unattested_packages: [package_name, ...]` +- `allow_provenance_publishers: [publisher_kind, ...]` +- `allow_unattested_publishers: [publisher_kind, ...]` as an accepted compatibility alias for `allow_provenance_publishers` + +`allow_provenance_publishers` is the canonical publisher override field. The parser also accepts `allow_unattested_publishers` as an alias when teams want a more explicit override-style name in review. Neither field treats missing attestations as trusted; they only constrain which attested publisher kinds count as verified provenance. + +## Version 2 supported rule ids + +Version `2` supports every version `1` rule id plus: + +- `missing_attestation` +- `unverified_provenance` +- `provenance_unavailable` +- `provenance_required` + +## Version 3 fields + +Version `3` supports every version `1` and `2` field plus: + +- `minimum_scorecard_score: float` + +`minimum_scorecard_score` is advisory by itself. It only affects policy outcomes when you also opt into the `scorecard_below_threshold` rule through `block_on`, `warn_on`, or `ignore_rules`. + +## Version 3 supported rule ids + +Version `3` supports every version `1` and `2` rule id plus: + +- `scorecard_below_threshold` + ## Semantics - `block_on` turns matching rule ids into blocking violations. @@ -37,8 +74,20 @@ The schema is intentionally conservative and fail-closed: - `max_added_packages` enforces a deterministic threshold on the added component count. - `allow_sources` enforces exact host matches against `source_url` hosts for added and changed components. - `ignore_rules` suppresses matching rule ids entirely. +- `missing_attestation` means PyPI release metadata was fetched successfully but no attestations were present. +- `provenance_unavailable` means the run did not have usable provenance evidence for that package, for example because enrichment was disabled, unsupported, or failed. +- `unverified_provenance` means attestations were present, but the provenance could not be verified against publisher metadata. +- `provenance_required` is a policy-only rule emitted when an explicit provenance requirement was not satisfied. +- `require_attestations_for_new_packages` applies only to added PyPI packages. +- `require_provenance_for_suspicious_sources` applies only when the component also triggered `suspicious_source`. +- `allow_unattested_packages` is a narrow package-name override for explicit missing-attestation exceptions only. +- `allow_unattested_packages` does not waive `provenance_unavailable` or `unverified_provenance`; those remain separate, reviewable policy decisions. +- `allow_provenance_publishers` and `allow_unattested_publishers` apply only when attestations exist and publisher kinds are available to verify. +- when enrichment is disabled, deterministic local mode is unchanged unless a provenance-aware policy explicitly turns unavailable evidence into a warning or block. +- `minimum_scorecard_score` does not create alerts or blocks on its own; it only becomes enforceable when `scorecard_below_threshold` is configured explicitly. +- Scorecard evidence remains an auxiliary trust signal. A high score is not proof of safety, and missing Scorecard data is not proof of risk. -## Example +## Version 1 example ```yaml version: 1 @@ -54,3 +103,29 @@ allow_sources: ignore_rules: - major_upgrade ``` + +## Version 2 example + +```yaml +version: 2 +block_on: + - provenance_required + - provenance_unavailable +warn_on: + - missing_attestation +require_attestations_for_new_packages: true +require_provenance_for_suspicious_sources: true +allow_unattested_packages: + - pip +allow_unattested_publishers: + - github actions +``` + +## Version 3 example + +```yaml +version: 3 +warn_on: + - scorecard_below_threshold +minimum_scorecard_score: 7.0 +``` diff --git a/tools/sbom-diff-and-risk/docs/self-provenance.md b/tools/sbom-diff-and-risk/docs/self-provenance.md index 96db8ee..a955ab2 100644 --- a/tools/sbom-diff-and-risk/docs/self-provenance.md +++ b/tools/sbom-diff-and-risk/docs/self-provenance.md @@ -11,7 +11,7 @@ The attested subjects are the exact Python distributables built from `tools/sbom Those two files are uploaded together as the workflow artifact named `sbom-diff-and-risk-dist`. The attestation applies to the built files themselves, not just to the artifact bundle name shown in the Actions UI. -Current attestations cover workflow-built wheel and sdist artifacts, not GitHub Release assets or PyPI-published distributions. +This repository does not currently publish PyPI Trusted Publishing provenance or immutable GitHub release attestations as part of this workflow. The current self-provenance coverage is limited to the workflow-produced wheel and source distribution files. ## Workflow and permissions diff --git a/tools/sbom-diff-and-risk/examples/policy-provenance-minimal.yml b/tools/sbom-diff-and-risk/examples/policy-provenance-minimal.yml new file mode 100644 index 0000000..00458ba --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/policy-provenance-minimal.yml @@ -0,0 +1,8 @@ +# Missing attestation remains a review signal, not proof of compromise. +version: 2 +warn_on: + - missing_attestation + - provenance_required +require_attestations_for_new_packages: true +allow_unattested_packages: + - pip diff --git a/tools/sbom-diff-and-risk/examples/policy-provenance-strict.yml b/tools/sbom-diff-and-risk/examples/policy-provenance-strict.yml new file mode 100644 index 0000000..9bf25e6 --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/policy-provenance-strict.yml @@ -0,0 +1,12 @@ +# Explicit provenance requirements for enriched PyPI evidence. +version: 2 +block_on: + - provenance_required + - provenance_unavailable + - unverified_provenance +warn_on: + - missing_attestation +require_attestations_for_new_packages: true +require_provenance_for_suspicious_sources: true +allow_unattested_publishers: + - github actions diff --git a/tools/sbom-diff-and-risk/examples/policy-scorecard-minimal.yml b/tools/sbom-diff-and-risk/examples/policy-scorecard-minimal.yml new file mode 100644 index 0000000..0eb6f91 --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/policy-scorecard-minimal.yml @@ -0,0 +1,4 @@ +version: 3 +warn_on: + - scorecard_below_threshold +minimum_scorecard_score: 7.0 diff --git a/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.json b/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.json index 3469a60..b0c5abe 100644 --- a/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.json +++ b/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.json @@ -475,8 +475,86 @@ "kind": "policy_check", "description": "Component source host was not present in the configured allow_sources list.", "finding_buckets": [] + }, + "missing_attestation": { + "rule_id": "missing_attestation", + "kind": "provenance_signal", + "description": "PyPI release metadata was fetched, but no attestations were published for the package release.", + "finding_buckets": [] + }, + "unverified_provenance": { + "rule_id": "unverified_provenance", + "kind": "provenance_signal", + "description": "PyPI attestations were present, but provenance could not be verified against publisher metadata.", + "finding_buckets": [] + }, + "provenance_unavailable": { + "rule_id": "provenance_unavailable", + "kind": "provenance_signal", + "description": "PyPI provenance evidence was unavailable because enrichment was disabled, unsupported, or errored.", + "finding_buckets": [] + }, + "provenance_required": { + "rule_id": "provenance_required", + "kind": "policy_check", + "description": "A configured provenance requirement was not satisfied for the component.", + "finding_buckets": [] + }, + "scorecard_below_threshold": { + "rule_id": "scorecard_below_threshold", + "kind": "policy_check", + "description": "A mapped repository's OpenSSF Scorecard score was below the configured minimum threshold.", + "finding_buckets": [] } }, + "provenance_summary": { + "components_in_scope": 2, + "pypi_components_in_scope": 2, + "pypi_components_without_provenance": 2, + "components_with_provenance": 0, + "components_with_attestations": 0, + "components_with_attestation_gaps": 0, + "components_with_enrichment_errors": 0, + "unsupported_components": 0 + }, + "attestation_summary": { + "files_evaluated": 0, + "files_with_attestations": 0, + "files_without_attestations": 0, + "packages_with_attestation_gaps": [], + "publisher_kind_counts": {} + }, + "scorecard_summary": { + "enabled": false, + "components_in_scope": 2, + "candidate_components": 0, + "supported_components": 0, + "components_with_mapped_repositories": 0, + "components_with_scorecards": 0, + "scorecard_unavailable": 0, + "repository_unmapped": 0, + "components_with_enrichment_errors": 0, + "results": [] + }, + "enrichment_metadata": { + "mode": "offline_default", + "pypi_enabled": false, + "pypi_timeout_seconds": null, + "pypi_network_access_performed": false, + "network_access_performed": false, + "candidate_components": 0, + "supported_components": 0, + "status_counts": {}, + "scorecard_enabled": false, + "scorecard_timeout_seconds": null, + "scorecard_network_access_performed": false, + "scorecard_candidate_components": 0, + "scorecard_supported_components": 0, + "scorecard_status_counts": {} + }, + "trust_signal_notes": [ + "PyPI components are present, but provenance enrichment was not enabled for this run." + ], "metadata": { "before_format": "cyclonedx-json", "after_format": "cyclonedx-json", @@ -555,6 +633,22 @@ "ignored_checks": 0 }, "exit_code": 1 + }, + "enrichment": { + "mode": "offline_default", + "pypi_enabled": false, + "pypi_timeout_seconds": null, + "pypi_network_access_performed": false, + "network_access_performed": false, + "candidate_components": 0, + "supported_components": 0, + "status_counts": {}, + "scorecard_enabled": false, + "scorecard_timeout_seconds": null, + "scorecard_network_access_performed": false, + "scorecard_candidate_components": 0, + "scorecard_supported_components": 0, + "scorecard_status_counts": {} } }, "notes": [ diff --git a/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.md b/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.md index f725fc7..4cb165a 100644 --- a/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.md +++ b/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.md @@ -24,6 +24,56 @@ - Warnings: 1 - Suppressed findings: 0 +## Provenance summary +- Enrichment mode: offline_default +- Network access performed: no +- Candidate components for enrichment: 0 +- Supported components for enrichment: 0 +- Observed provenance status counts: none +- Components in scope: 2 +- PyPI components in scope: 2 +- PyPI components without provenance records: 2 +- Components with provenance evidence: 0 +- Components with attestations: 0 +- Components with attestation gaps: 0 +- Components with enrichment errors: 0 +- Unsupported components: 0 + +## Attestation gaps +| component | version | statuses | +|-----------|---------|----------| +| _none_ | | | + +## Policy impact for provenance-related rules +| rule id | component | level | message | +|---------|-----------|-------|---------| +| _none_ | | | | + +## Trust signal notes +- PyPI components are present, but provenance enrichment was not enabled for this run. + +## Scorecard summary +- Enrichment enabled: no +- Network access performed: no +- Candidate components for Scorecard enrichment: 0 +- Components with supported repository mappings: 0 +- Components with mapped repositories: 0 +- Components with available Scorecards: 0 +- Scorecard unavailable: 0 +- Repository unmapped: 0 +- Components with enrichment errors: 0 +- Observed Scorecard status counts: none + +## Scorecard results +| component | version | repository | score | status | +|-----------|---------|------------|-------|--------| +| _none_ | | | | | + +## Policy impact for Scorecard-related rules +| rule id | component | level | message | +|---------|-----------|-------|---------| +| _none_ | | | | + ## Added components | name | version | ecosystem | risk buckets | |------|---------|-----------|--------------| diff --git a/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.json b/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.json index ba0d7c7..4a8ce07 100644 --- a/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.json +++ b/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.json @@ -410,8 +410,86 @@ "kind": "policy_check", "description": "Component source host was not present in the configured allow_sources list.", "finding_buckets": [] + }, + "missing_attestation": { + "rule_id": "missing_attestation", + "kind": "provenance_signal", + "description": "PyPI release metadata was fetched, but no attestations were published for the package release.", + "finding_buckets": [] + }, + "unverified_provenance": { + "rule_id": "unverified_provenance", + "kind": "provenance_signal", + "description": "PyPI attestations were present, but provenance could not be verified against publisher metadata.", + "finding_buckets": [] + }, + "provenance_unavailable": { + "rule_id": "provenance_unavailable", + "kind": "provenance_signal", + "description": "PyPI provenance evidence was unavailable because enrichment was disabled, unsupported, or errored.", + "finding_buckets": [] + }, + "provenance_required": { + "rule_id": "provenance_required", + "kind": "policy_check", + "description": "A configured provenance requirement was not satisfied for the component.", + "finding_buckets": [] + }, + "scorecard_below_threshold": { + "rule_id": "scorecard_below_threshold", + "kind": "policy_check", + "description": "A mapped repository's OpenSSF Scorecard score was below the configured minimum threshold.", + "finding_buckets": [] } }, + "provenance_summary": { + "components_in_scope": 2, + "pypi_components_in_scope": 2, + "pypi_components_without_provenance": 2, + "components_with_provenance": 0, + "components_with_attestations": 0, + "components_with_attestation_gaps": 0, + "components_with_enrichment_errors": 0, + "unsupported_components": 0 + }, + "attestation_summary": { + "files_evaluated": 0, + "files_with_attestations": 0, + "files_without_attestations": 0, + "packages_with_attestation_gaps": [], + "publisher_kind_counts": {} + }, + "scorecard_summary": { + "enabled": false, + "components_in_scope": 2, + "candidate_components": 0, + "supported_components": 0, + "components_with_mapped_repositories": 0, + "components_with_scorecards": 0, + "scorecard_unavailable": 0, + "repository_unmapped": 0, + "components_with_enrichment_errors": 0, + "results": [] + }, + "enrichment_metadata": { + "mode": "offline_default", + "pypi_enabled": false, + "pypi_timeout_seconds": null, + "pypi_network_access_performed": false, + "network_access_performed": false, + "candidate_components": 0, + "supported_components": 0, + "status_counts": {}, + "scorecard_enabled": false, + "scorecard_timeout_seconds": null, + "scorecard_network_access_performed": false, + "scorecard_candidate_components": 0, + "scorecard_supported_components": 0, + "scorecard_status_counts": {} + }, + "trust_signal_notes": [ + "PyPI components are present, but provenance enrichment was not enabled for this run." + ], "metadata": { "before_format": "cyclonedx-json", "after_format": "cyclonedx-json", @@ -453,6 +531,22 @@ "ignored_checks": 0 }, "exit_code": 0 + }, + "enrichment": { + "mode": "offline_default", + "pypi_enabled": false, + "pypi_timeout_seconds": null, + "pypi_network_access_performed": false, + "network_access_performed": false, + "candidate_components": 0, + "supported_components": 0, + "status_counts": {}, + "scorecard_enabled": false, + "scorecard_timeout_seconds": null, + "scorecard_network_access_performed": false, + "scorecard_candidate_components": 0, + "scorecard_supported_components": 0, + "scorecard_status_counts": {} } }, "notes": [ diff --git a/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.md b/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.md index d4ce8d1..1bf7a6d 100644 --- a/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.md +++ b/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.md @@ -24,6 +24,56 @@ - Warnings: 1 - Suppressed findings: 0 +## Provenance summary +- Enrichment mode: offline_default +- Network access performed: no +- Candidate components for enrichment: 0 +- Supported components for enrichment: 0 +- Observed provenance status counts: none +- Components in scope: 2 +- PyPI components in scope: 2 +- PyPI components without provenance records: 2 +- Components with provenance evidence: 0 +- Components with attestations: 0 +- Components with attestation gaps: 0 +- Components with enrichment errors: 0 +- Unsupported components: 0 + +## Attestation gaps +| component | version | statuses | +|-----------|---------|----------| +| _none_ | | | + +## Policy impact for provenance-related rules +| rule id | component | level | message | +|---------|-----------|-------|---------| +| _none_ | | | | + +## Trust signal notes +- PyPI components are present, but provenance enrichment was not enabled for this run. + +## Scorecard summary +- Enrichment enabled: no +- Network access performed: no +- Candidate components for Scorecard enrichment: 0 +- Components with supported repository mappings: 0 +- Components with mapped repositories: 0 +- Components with available Scorecards: 0 +- Scorecard unavailable: 0 +- Repository unmapped: 0 +- Components with enrichment errors: 0 +- Observed Scorecard status counts: none + +## Scorecard results +| component | version | repository | score | status | +|-----------|---------|------------|-------|--------| +| _none_ | | | | | + +## Policy impact for Scorecard-related rules +| rule id | component | level | message | +|---------|-----------|-------|---------| +| _none_ | | | | + ## Added components | name | version | ecosystem | risk buckets | |------|---------|-----------|--------------| diff --git a/tools/sbom-diff-and-risk/examples/sample-provenance-report.json b/tools/sbom-diff-and-risk/examples/sample-provenance-report.json new file mode 100644 index 0000000..ebab5af --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/sample-provenance-report.json @@ -0,0 +1,792 @@ +{ + "summary": { + "added": 2, + "removed": 0, + "changed": 1, + "risk_counts": { + "new_package": 2, + "major_upgrade": 0, + "version_change_unclassified": 1, + "unknown_license": 0, + "stale_package": 0, + "suspicious_source": 0, + "not_evaluated": 0 + } + }, + "components": { + "added": [ + { + "name": "urllib3", + "version": "2.2.1", + "ecosystem": "pypi", + "purl": "pkg:pypi/urllib3@2.2.1", + "license_id": null, + "supplier": null, + "source_url": "https://pypi.org/project/urllib3/2.2.1/", + "bom_ref": null, + "raw_type": null, + "evidence": { + "provenance": { + "provider": "pypi", + "requested": true, + "supported": true, + "lookup_performed": true, + "package_name": "urllib3", + "package_version": "2.2.1", + "release_url": "https://pypi.org/project/urllib3/2.2.1/", + "statuses": [ + "provenance_available", + "attestation_available" + ], + "files": [ + { + "filename": "urllib3-2.2.1.tar.gz", + "url": "https://files.pythonhosted.org/packages/urllib3-2.2.1.tar.gz", + "sha256": "deadbeef", + "upload_time": "2026-04-01T00:00:00.000000Z", + "yanked": false, + "statuses": [ + "provenance_available", + "attestation_available" + ], + "attestation_count": 1, + "predicate_types": [ + "https://example.test/attestation/v1" + ], + "publisher_kinds": [ + "github actions" + ], + "error": null + } + ], + "files_evaluated": 1, + "files_with_attestations": 1, + "files_without_attestations": 0, + "error": null + } + } + }, + { + "name": "mystery-lib", + "version": "1.0.0", + "ecosystem": "pypi", + "purl": "pkg:pypi/mystery-lib@1.0.0", + "license_id": null, + "supplier": null, + "source_url": "https://pypi.org/project/mystery-lib/1.0.0/", + "bom_ref": null, + "raw_type": null, + "evidence": { + "provenance": { + "provider": "pypi", + "requested": true, + "supported": true, + "lookup_performed": true, + "package_name": "mystery-lib", + "package_version": "1.0.0", + "release_url": "https://pypi.org/project/mystery-lib/1.0.0/", + "statuses": [ + "attestation_unavailable" + ], + "files": [ + { + "filename": "mystery-lib-1.0.0.tar.gz", + "url": "https://files.pythonhosted.org/packages/mystery-lib-1.0.0.tar.gz", + "sha256": "deadbeef", + "upload_time": "2026-04-01T00:00:00.000000Z", + "yanked": false, + "statuses": [ + "attestation_unavailable" + ], + "attestation_count": 0, + "predicate_types": [], + "publisher_kinds": [], + "error": null + } + ], + "files_evaluated": 1, + "files_with_attestations": 0, + "files_without_attestations": 1, + "error": null + } + } + } + ], + "removed": [], + "changed": [ + { + "key": "purl:pkg:pypi/legacy-lib", + "classification": "version_changed", + "before": { + "name": "legacy-lib", + "version": "1.0.0", + "ecosystem": "pypi", + "purl": "pkg:pypi/legacy-lib@1.0.0", + "license_id": null, + "supplier": null, + "source_url": null, + "bom_ref": null, + "raw_type": null, + "evidence": {} + }, + "after": { + "name": "legacy-lib", + "version": "1.1.0", + "ecosystem": "pypi", + "purl": "pkg:pypi/legacy-lib@1.1.0", + "license_id": null, + "supplier": null, + "source_url": null, + "bom_ref": null, + "raw_type": null, + "evidence": { + "provenance": { + "provider": "pypi", + "requested": true, + "supported": true, + "lookup_performed": true, + "package_name": "legacy-lib", + "package_version": "1.1.0", + "release_url": "https://pypi.org/project/legacy-lib/1.1.0/", + "statuses": [ + "provenance_available", + "attestation_available" + ], + "files": [ + { + "filename": "legacy-lib-1.1.0.tar.gz", + "url": "https://files.pythonhosted.org/packages/legacy-lib-1.1.0.tar.gz", + "sha256": "deadbeef", + "upload_time": "2026-04-01T00:00:00.000000Z", + "yanked": false, + "statuses": [ + "provenance_available", + "attestation_available" + ], + "attestation_count": 1, + "predicate_types": [ + "https://example.test/attestation/v1" + ], + "publisher_kinds": [ + "manual upload" + ], + "error": null + } + ], + "files_evaluated": 1, + "files_with_attestations": 1, + "files_without_attestations": 0, + "error": null + } + } + } + } + ] + }, + "risks": [ + { + "bucket": "new_package", + "component_key": "purl:pkg:pypi/urllib3", + "component": { + "name": "urllib3", + "version": "2.2.1", + "ecosystem": "pypi", + "purl": "pkg:pypi/urllib3@2.2.1", + "license_id": null, + "supplier": null, + "source_url": "https://pypi.org/project/urllib3/2.2.1/", + "bom_ref": null, + "raw_type": null, + "evidence": { + "provenance": { + "provider": "pypi", + "requested": true, + "supported": true, + "lookup_performed": true, + "package_name": "urllib3", + "package_version": "2.2.1", + "release_url": "https://pypi.org/project/urllib3/2.2.1/", + "statuses": [ + "provenance_available", + "attestation_available" + ], + "files": [ + { + "filename": "urllib3-2.2.1.tar.gz", + "url": "https://files.pythonhosted.org/packages/urllib3-2.2.1.tar.gz", + "sha256": "deadbeef", + "upload_time": "2026-04-01T00:00:00.000000Z", + "yanked": false, + "statuses": [ + "provenance_available", + "attestation_available" + ], + "attestation_count": 1, + "predicate_types": [ + "https://example.test/attestation/v1" + ], + "publisher_kinds": [ + "github actions" + ], + "error": null + } + ], + "files_evaluated": 1, + "files_with_attestations": 1, + "files_without_attestations": 0, + "error": null + } + } + }, + "rationale": "Component was not present in the before input." + }, + { + "bucket": "new_package", + "component_key": "purl:pkg:pypi/mystery-lib", + "component": { + "name": "mystery-lib", + "version": "1.0.0", + "ecosystem": "pypi", + "purl": "pkg:pypi/mystery-lib@1.0.0", + "license_id": null, + "supplier": null, + "source_url": "https://pypi.org/project/mystery-lib/1.0.0/", + "bom_ref": null, + "raw_type": null, + "evidence": { + "provenance": { + "provider": "pypi", + "requested": true, + "supported": true, + "lookup_performed": true, + "package_name": "mystery-lib", + "package_version": "1.0.0", + "release_url": "https://pypi.org/project/mystery-lib/1.0.0/", + "statuses": [ + "attestation_unavailable" + ], + "files": [ + { + "filename": "mystery-lib-1.0.0.tar.gz", + "url": "https://files.pythonhosted.org/packages/mystery-lib-1.0.0.tar.gz", + "sha256": "deadbeef", + "upload_time": "2026-04-01T00:00:00.000000Z", + "yanked": false, + "statuses": [ + "attestation_unavailable" + ], + "attestation_count": 0, + "predicate_types": [], + "publisher_kinds": [], + "error": null + } + ], + "files_evaluated": 1, + "files_with_attestations": 0, + "files_without_attestations": 1, + "error": null + } + } + }, + "rationale": "Component was not present in the before input." + }, + { + "bucket": "version_change_unclassified", + "component_key": "purl:pkg:pypi/legacy-lib", + "component": { + "name": "legacy-lib", + "version": "1.1.0", + "ecosystem": "pypi", + "purl": "pkg:pypi/legacy-lib@1.1.0", + "license_id": null, + "supplier": null, + "source_url": null, + "bom_ref": null, + "raw_type": null, + "evidence": { + "provenance": { + "provider": "pypi", + "requested": true, + "supported": true, + "lookup_performed": true, + "package_name": "legacy-lib", + "package_version": "1.1.0", + "release_url": "https://pypi.org/project/legacy-lib/1.1.0/", + "statuses": [ + "provenance_available", + "attestation_available" + ], + "files": [ + { + "filename": "legacy-lib-1.1.0.tar.gz", + "url": "https://files.pythonhosted.org/packages/legacy-lib-1.1.0.tar.gz", + "sha256": "deadbeef", + "upload_time": "2026-04-01T00:00:00.000000Z", + "yanked": false, + "statuses": [ + "provenance_available", + "attestation_available" + ], + "attestation_count": 1, + "predicate_types": [ + "https://example.test/attestation/v1" + ], + "publisher_kinds": [ + "manual upload" + ], + "error": null + } + ], + "files_evaluated": 1, + "files_with_attestations": 1, + "files_without_attestations": 0, + "error": null + } + } + }, + "rationale": "Version changed but did not qualify as a parseable SemVer major upgrade." + } + ], + "policy_evaluation": { + "applied": true, + "policy_path": "examples/policy-provenance-strict.yml", + "effective_policy": { + "version": 2, + "block_on": [ + "provenance_required", + "unverified_provenance" + ], + "warn_on": [ + "missing_attestation" + ], + "max_added_packages": null, + "allow_sources": [], + "ignore_rules": [], + "require_attestations_for_new_packages": true, + "require_provenance_for_suspicious_sources": false, + "allow_unattested_packages": [], + "allow_provenance_publishers": [ + "github actions" + ] + }, + "blocking_violations": [ + { + "rule_id": "provenance_required", + "level": "block", + "message": "Provenance is required for new package, but no attestations were published for this PyPI package.", + "component_key": "purl:pkg:pypi/mystery-lib", + "component_name": "mystery-lib", + "finding_bucket": null, + "suppression_reason": null + }, + { + "rule_id": "unverified_provenance", + "level": "block", + "message": "PyPI attestations were present, but publisher kinds manual upload did not match allow_provenance_publishers=github actions.", + "component_key": "purl:pkg:pypi/legacy-lib", + "component_name": "legacy-lib", + "finding_bucket": null, + "suppression_reason": null + } + ], + "warning_violations": [ + { + "rule_id": "missing_attestation", + "level": "warn", + "message": "PyPI release metadata was fetched, but no attestations were published for this package release.", + "component_key": "purl:pkg:pypi/mystery-lib", + "component_name": "mystery-lib", + "finding_bucket": null, + "suppression_reason": null + } + ], + "suppressed_violations": [], + "totals": { + "blocking": 2, + "warning": 1, + "suppressed": 0, + "ignored_checks": 0 + }, + "exit_code": 1 + }, + "blocking_findings": [ + { + "rule_id": "provenance_required", + "level": "block", + "message": "Provenance is required for new package, but no attestations were published for this PyPI package.", + "component_key": "purl:pkg:pypi/mystery-lib", + "component_name": "mystery-lib", + "finding_bucket": null, + "suppression_reason": null + }, + { + "rule_id": "unverified_provenance", + "level": "block", + "message": "PyPI attestations were present, but publisher kinds manual upload did not match allow_provenance_publishers=github actions.", + "component_key": "purl:pkg:pypi/legacy-lib", + "component_name": "legacy-lib", + "finding_bucket": null, + "suppression_reason": null + } + ], + "warning_findings": [ + { + "rule_id": "missing_attestation", + "level": "warn", + "message": "PyPI release metadata was fetched, but no attestations were published for this package release.", + "component_key": "purl:pkg:pypi/mystery-lib", + "component_name": "mystery-lib", + "finding_bucket": null, + "suppression_reason": null + } + ], + "suppressed_findings": [], + "rule_catalog": { + "new_package": { + "rule_id": "new_package", + "kind": "risk_finding", + "description": "Component is present only in the after input.", + "finding_buckets": [ + "new_package" + ] + }, + "major_upgrade": { + "rule_id": "major_upgrade", + "kind": "risk_finding", + "description": "Version change is a parseable SemVer major upgrade.", + "finding_buckets": [ + "major_upgrade" + ] + }, + "version_change_unclassified": { + "rule_id": "version_change_unclassified", + "kind": "risk_finding", + "description": "Version changed but could not be classified as a reliable major SemVer upgrade.", + "finding_buckets": [ + "version_change_unclassified" + ] + }, + "unknown_license": { + "rule_id": "unknown_license", + "kind": "risk_finding", + "description": "License metadata is missing, empty, UNKNOWN, or NOASSERTION.", + "finding_buckets": [ + "unknown_license" + ] + }, + "suspicious_source": { + "rule_id": "suspicious_source", + "kind": "risk_finding", + "description": "Source provenance is missing or points to a suspicious scheme, path, or host.", + "finding_buckets": [ + "suspicious_source" + ] + }, + "stale_package": { + "rule_id": "stale_package", + "kind": "risk_finding", + "description": "Staleness check result. Offline mode maps this rule to not_evaluated instead of guessing.", + "finding_buckets": [ + "stale_package", + "not_evaluated" + ] + }, + "max_added_packages": { + "rule_id": "max_added_packages", + "kind": "policy_check", + "description": "Added package count exceeded the configured deterministic threshold.", + "finding_buckets": [] + }, + "allow_sources": { + "rule_id": "allow_sources", + "kind": "policy_check", + "description": "Component source host was not present in the configured allow_sources list.", + "finding_buckets": [] + }, + "missing_attestation": { + "rule_id": "missing_attestation", + "kind": "provenance_signal", + "description": "PyPI release metadata was fetched, but no attestations were published for the package release.", + "finding_buckets": [] + }, + "unverified_provenance": { + "rule_id": "unverified_provenance", + "kind": "provenance_signal", + "description": "PyPI attestations were present, but provenance could not be verified against publisher metadata.", + "finding_buckets": [] + }, + "provenance_unavailable": { + "rule_id": "provenance_unavailable", + "kind": "provenance_signal", + "description": "PyPI provenance evidence was unavailable because enrichment was disabled, unsupported, or errored.", + "finding_buckets": [] + }, + "provenance_required": { + "rule_id": "provenance_required", + "kind": "policy_check", + "description": "A configured provenance requirement was not satisfied for the component.", + "finding_buckets": [] + }, + "scorecard_below_threshold": { + "rule_id": "scorecard_below_threshold", + "kind": "policy_check", + "description": "A mapped repository's OpenSSF Scorecard score was below the configured minimum threshold.", + "finding_buckets": [] + } + }, + "provenance_summary": { + "components_in_scope": 3, + "pypi_components_in_scope": 3, + "pypi_components_without_provenance": 0, + "components_with_provenance": 2, + "components_with_attestations": 2, + "components_with_attestation_gaps": 1, + "components_with_enrichment_errors": 0, + "unsupported_components": 0 + }, + "attestation_summary": { + "files_evaluated": 3, + "files_with_attestations": 2, + "files_without_attestations": 1, + "packages_with_attestation_gaps": [ + { + "component_key": "purl:pkg:pypi/mystery-lib", + "name": "mystery-lib", + "version": "1.0.0", + "statuses": [ + "attestation_unavailable" + ] + } + ], + "publisher_kind_counts": { + "github actions": 1, + "manual upload": 1 + } + }, + "scorecard_summary": { + "enabled": false, + "components_in_scope": 3, + "candidate_components": 0, + "supported_components": 0, + "components_with_mapped_repositories": 0, + "components_with_scorecards": 0, + "scorecard_unavailable": 0, + "repository_unmapped": 0, + "components_with_enrichment_errors": 0, + "results": [] + }, + "enrichment_metadata": { + "mode": "opt_in_pypi", + "pypi_enabled": true, + "pypi_timeout_seconds": 3.0, + "pypi_network_access_performed": true, + "network_access_performed": true, + "candidate_components": 3, + "supported_components": 3, + "status_counts": { + "attestation_available": 2, + "attestation_unavailable": 1, + "provenance_available": 2 + }, + "scorecard_enabled": false, + "scorecard_timeout_seconds": null, + "scorecard_network_access_performed": false, + "scorecard_candidate_components": 0, + "scorecard_supported_components": 0, + "scorecard_status_counts": {} + }, + "trust_signal_notes": [ + "Missing attestations indicate an attestation gap for the release; they are not treated as proof of compromise.", + "Observed attestation publisher kinds: github actions, manual upload.", + "Policy produced 3 provenance-related blocking or warning decision(s)." + ], + "metadata": { + "before_format": "requirements-txt", + "after_format": "requirements-txt", + "generated_at": null, + "strict": false, + "stub": false, + "policy_evaluation": { + "applied": true, + "policy_path": "examples/policy-provenance-strict.yml", + "effective_policy": { + "version": 2, + "block_on": [ + "provenance_required", + "unverified_provenance" + ], + "warn_on": [ + "missing_attestation" + ], + "max_added_packages": null, + "allow_sources": [], + "ignore_rules": [], + "require_attestations_for_new_packages": true, + "require_provenance_for_suspicious_sources": false, + "allow_unattested_packages": [], + "allow_provenance_publishers": [ + "github actions" + ] + }, + "blocking_violations": [ + { + "rule_id": "provenance_required", + "level": "block", + "message": "Provenance is required for new package, but no attestations were published for this PyPI package.", + "component_key": "purl:pkg:pypi/mystery-lib", + "component_name": "mystery-lib", + "finding_bucket": null, + "suppression_reason": null + }, + { + "rule_id": "unverified_provenance", + "level": "block", + "message": "PyPI attestations were present, but publisher kinds manual upload did not match allow_provenance_publishers=github actions.", + "component_key": "purl:pkg:pypi/legacy-lib", + "component_name": "legacy-lib", + "finding_bucket": null, + "suppression_reason": null + } + ], + "warning_violations": [ + { + "rule_id": "missing_attestation", + "level": "warn", + "message": "PyPI release metadata was fetched, but no attestations were published for this package release.", + "component_key": "purl:pkg:pypi/mystery-lib", + "component_name": "mystery-lib", + "finding_bucket": null, + "suppression_reason": null + } + ], + "suppressed_violations": [], + "totals": { + "blocking": 2, + "warning": 1, + "suppressed": 0, + "ignored_checks": 0 + }, + "exit_code": 1 + }, + "enrichment": { + "mode": "opt_in_pypi", + "pypi_enabled": true, + "pypi_timeout_seconds": 3.0, + "pypi_network_access_performed": true, + "network_access_performed": true, + "candidate_components": 3, + "supported_components": 3, + "status_counts": { + "attestation_available": 2, + "attestation_unavailable": 1, + "provenance_available": 2 + }, + "scorecard_enabled": false, + "scorecard_timeout_seconds": null, + "scorecard_network_access_performed": false, + "scorecard_candidate_components": 0, + "scorecard_supported_components": 0, + "scorecard_status_counts": {} + } + }, + "notes": [ + "This tool uses heuristic risk classification.", + "PyPI provenance enrichment was requested explicitly." + ], + "provenance_policy": { + "configured": true, + "requirements": { + "require_attestations_for_new_packages": true, + "require_provenance_for_suspicious_sources": false, + "allow_unattested_packages": [], + "allow_provenance_publishers": [ + "github actions" + ] + }, + "counts": { + "blocking": 2, + "warning": 1, + "suppressed": 0 + }, + "blocking": [ + { + "rule_id": "provenance_required", + "level": "block", + "message": "Provenance is required for new package, but no attestations were published for this PyPI package.", + "component_key": "purl:pkg:pypi/mystery-lib", + "component_name": "mystery-lib", + "finding_bucket": null, + "suppression_reason": null + }, + { + "rule_id": "unverified_provenance", + "level": "block", + "message": "PyPI attestations were present, but publisher kinds manual upload did not match allow_provenance_publishers=github actions.", + "component_key": "purl:pkg:pypi/legacy-lib", + "component_name": "legacy-lib", + "finding_bucket": null, + "suppression_reason": null + } + ], + "warning": [ + { + "rule_id": "missing_attestation", + "level": "warn", + "message": "PyPI release metadata was fetched, but no attestations were published for this package release.", + "component_key": "purl:pkg:pypi/mystery-lib", + "component_name": "mystery-lib", + "finding_bucket": null, + "suppression_reason": null + } + ], + "suppressed": [] + }, + "provenance_policy_impact": { + "configured": true, + "requirements": { + "require_attestations_for_new_packages": true, + "require_provenance_for_suspicious_sources": false, + "allow_unattested_packages": [], + "allow_provenance_publishers": [ + "github actions" + ] + }, + "counts": { + "blocking": 2, + "warning": 1, + "suppressed": 0 + }, + "blocking": [ + { + "rule_id": "provenance_required", + "level": "block", + "message": "Provenance is required for new package, but no attestations were published for this PyPI package.", + "component_key": "purl:pkg:pypi/mystery-lib", + "component_name": "mystery-lib", + "finding_bucket": null, + "suppression_reason": null + }, + { + "rule_id": "unverified_provenance", + "level": "block", + "message": "PyPI attestations were present, but publisher kinds manual upload did not match allow_provenance_publishers=github actions.", + "component_key": "purl:pkg:pypi/legacy-lib", + "component_name": "legacy-lib", + "finding_bucket": null, + "suppression_reason": null + } + ], + "warning": [ + { + "rule_id": "missing_attestation", + "level": "warn", + "message": "PyPI release metadata was fetched, but no attestations were published for this package release.", + "component_key": "purl:pkg:pypi/mystery-lib", + "component_name": "mystery-lib", + "finding_bucket": null, + "suppression_reason": null + } + ], + "suppressed": [] + } +} diff --git a/tools/sbom-diff-and-risk/examples/sample-provenance-report.md b/tools/sbom-diff-and-risk/examples/sample-provenance-report.md new file mode 100644 index 0000000..963c7ce --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/sample-provenance-report.md @@ -0,0 +1,123 @@ +# sbom-diff-and-risk report + +## Summary +- Before format: requirements-txt +- After format: requirements-txt +- Added: 2 +- Removed: 0 +- Version changes: 1 + +## Risk buckets +- new_package: 2 +- major_upgrade: 0 +- version_change_unclassified: 1 +- unknown_license: 0 +- stale_package: 0 +- suspicious_source: 0 +- not_evaluated: 0 + +## Policy summary +- Applied: yes +- Policy path: examples/policy-provenance-strict.yml +- Exit code: 1 +- Blocking findings: 2 +- Warnings: 1 +- Suppressed findings: 0 + +## Provenance summary +- Enrichment mode: opt_in_pypi +- Network access performed: yes +- Candidate components for enrichment: 3 +- Supported components for enrichment: 3 +- Observed provenance status counts: attestation_available=2, attestation_unavailable=1, provenance_available=2 +- Components in scope: 3 +- PyPI components in scope: 3 +- PyPI components without provenance records: 0 +- Components with provenance evidence: 2 +- Components with attestations: 2 +- Components with attestation gaps: 1 +- Components with enrichment errors: 0 +- Unsupported components: 0 + +## Attestation gaps +| component | version | statuses | +|-----------|---------|----------| +| mystery-lib | 1.0.0 | attestation_unavailable | + +## Policy impact for provenance-related rules +- Configured provenance policy: yes +- Require attestations for new packages: yes +- Require provenance for suspicious sources: no +- Allow unattested packages: none +- Allowed provenance publishers: github actions +- Provenance policy decisions: blocking=2, warning=1, suppressed=0 +| rule id | component | level | message | +|---------|-----------|-------|---------| +| provenance_required | mystery-lib | block | Provenance is required for new package, but no attestations were published for this PyPI package. | +| unverified_provenance | legacy-lib | block | PyPI attestations were present, but publisher kinds manual upload did not match allow_provenance_publishers=github actions. | +| missing_attestation | mystery-lib | warn | PyPI release metadata was fetched, but no attestations were published for this package release. | + +## Trust signal notes +- Missing attestations indicate an attestation gap for the release; they are not treated as proof of compromise. +- Observed attestation publisher kinds: github actions, manual upload. +- Policy produced 3 provenance-related blocking or warning decision(s). + +## Scorecard summary +- Enrichment enabled: no +- Network access performed: no +- Candidate components for Scorecard enrichment: 0 +- Components with supported repository mappings: 0 +- Components with mapped repositories: 0 +- Components with available Scorecards: 0 +- Scorecard unavailable: 0 +- Repository unmapped: 0 +- Components with enrichment errors: 0 +- Observed Scorecard status counts: none + +## Scorecard results +| component | version | repository | score | status | +|-----------|---------|------------|-------|--------| +| _none_ | | | | | + +## Policy impact for Scorecard-related rules +| rule id | component | level | message | +|---------|-----------|-------|---------| +| _none_ | | | | + +## Added components +| name | version | ecosystem | risk buckets | +|------|---------|-----------|--------------| +| urllib3 | 2.2.1 | pypi | new_package | +| mystery-lib | 1.0.0 | pypi | new_package | + +## Removed components +| name | version | ecosystem | +|------|---------|-----------| +| _none_ | | | + +## Version changes +| name | before | after | classification | risk buckets | +|------|--------|-------|----------------|--------------| +| legacy-lib | 1.0.0 | 1.1.0 | version_changed | version_change_unclassified | + +## Risk findings +| bucket | component | version | rationale | +|--------|-----------|---------|-----------| +| new_package | urllib3 | 2.2.1 | Component was not present in the before input. | +| new_package | mystery-lib | 1.0.0 | Component was not present in the before input. | +| version_change_unclassified | legacy-lib | 1.1.0 | Version changed but did not qualify as a parseable SemVer major upgrade. | + +## Blocking violations +| rule id | component | level | message | +|---------|-----------|-------|---------| +| provenance_required | mystery-lib | block | Provenance is required for new package, but no attestations were published for this PyPI package. | +| unverified_provenance | legacy-lib | block | PyPI attestations were present, but publisher kinds manual upload did not match allow_provenance_publishers=github actions. | + +## Warnings +| rule id | component | level | message | +|---------|-----------|-------|---------| +| missing_attestation | mystery-lib | warn | PyPI release metadata was fetched, but no attestations were published for this package release. | + +## Notes +- This tool uses heuristic risk classification. +- PyPI provenance enrichment was requested explicitly. diff --git a/tools/sbom-diff-and-risk/examples/sample-provenance-report.sarif b/tools/sbom-diff-and-risk/examples/sample-provenance-report.sarif new file mode 100644 index 0000000..6144888 --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/sample-provenance-report.sarif @@ -0,0 +1,150 @@ +{ + "$schema": "https://json.schemastore.org/sarif-2.1.0.json", + "version": "2.1.0", + "runs": [ + { + "tool": { + "driver": { + "name": "sbom-diff-risk", + "fullName": "sbom-diff-risk", + "version": "0.3.0", + "semanticVersion": "0.3.0", + "rules": [ + { + "id": "sdr.policy_violation.provenance_required", + "name": "policy_violation.provenance_required", + "shortDescription": { + "text": "Policy violation: provenance_required" + }, + "fullDescription": { + "text": "A configured provenance requirement was not satisfied for the component." + }, + "defaultConfiguration": { + "level": "error" + }, + "properties": { + "tags": [ + "supply-chain", + "policy", + "provenance" + ] + } + }, + { + "id": "sdr.policy_violation.unverified_provenance", + "name": "policy_violation.unverified_provenance", + "shortDescription": { + "text": "Policy violation: unverified_provenance" + }, + "fullDescription": { + "text": "PyPI attestations were present, but provenance could not be verified against publisher metadata." + }, + "defaultConfiguration": { + "level": "error" + }, + "properties": { + "tags": [ + "supply-chain", + "policy", + "provenance" + ] + } + } + ] + } + }, + "artifacts": [ + { + "location": { + "uri": "examples/requirements_before.txt", + "uriBaseId": "%SRCROOT%" + } + }, + { + "location": { + "uri": "examples/requirements_after.txt", + "uriBaseId": "%SRCROOT%" + } + } + ], + "properties": { + "sbom_diff_risk": { + "result_limit": 5000, + "total_candidate_results": 2, + "emitted_results": 2, + "omitted_results": 0, + "truncated": false, + "prioritization": "error results first, then warning, then note; direct mapped findings before policy-only checks; stable rule priority and component key tie-breakers.", + "warning": null + } + }, + "results": [ + { + "ruleId": "sdr.policy_violation.provenance_required", + "level": "error", + "message": { + "text": "mystery-lib: Provenance required for new package; no attestations were published." + }, + "locations": [ + { + "physicalLocation": { + "artifactLocation": { + "uri": "examples/requirements_after.txt", + "uriBaseId": "%SRCROOT%" + }, + "region": { + "startLine": 1 + } + } + } + ], + "partialFingerprints": { + "ruleId": "sdr.policy_violation.provenance_required", + "componentKey": "purl:pkg:pypi/mystery-lib" + }, + "properties": { + "policy_rule_id": "provenance_required", + "component_key": "purl:pkg:pypi/mystery-lib", + "component_name": "mystery-lib", + "result_kind": "policy_violation" + } + }, + { + "ruleId": "sdr.policy_violation.unverified_provenance", + "level": "error", + "message": { + "text": "legacy-lib: PyPI attestation publisher could not be verified by policy." + }, + "locations": [ + { + "physicalLocation": { + "artifactLocation": { + "uri": "examples/requirements_after.txt", + "uriBaseId": "%SRCROOT%" + }, + "region": { + "startLine": 1 + } + } + } + ], + "partialFingerprints": { + "ruleId": "sdr.policy_violation.unverified_provenance", + "componentKey": "purl:pkg:pypi/legacy-lib" + }, + "properties": { + "policy_rule_id": "unverified_provenance", + "component_key": "purl:pkg:pypi/legacy-lib", + "component_name": "legacy-lib", + "result_kind": "policy_violation" + } + } + ], + "originalUriBaseIds": { + "%SRCROOT%": { + "uri": "file:///__PROJECT_ROOT__/" + } + } + } + ] +} diff --git a/tools/sbom-diff-and-risk/examples/sample-report.json b/tools/sbom-diff-and-risk/examples/sample-report.json index 6c4f73a..39a0230 100644 --- a/tools/sbom-diff-and-risk/examples/sample-report.json +++ b/tools/sbom-diff-and-risk/examples/sample-report.json @@ -379,8 +379,86 @@ "kind": "policy_check", "description": "Component source host was not present in the configured allow_sources list.", "finding_buckets": [] + }, + "missing_attestation": { + "rule_id": "missing_attestation", + "kind": "provenance_signal", + "description": "PyPI release metadata was fetched, but no attestations were published for the package release.", + "finding_buckets": [] + }, + "unverified_provenance": { + "rule_id": "unverified_provenance", + "kind": "provenance_signal", + "description": "PyPI attestations were present, but provenance could not be verified against publisher metadata.", + "finding_buckets": [] + }, + "provenance_unavailable": { + "rule_id": "provenance_unavailable", + "kind": "provenance_signal", + "description": "PyPI provenance evidence was unavailable because enrichment was disabled, unsupported, or errored.", + "finding_buckets": [] + }, + "provenance_required": { + "rule_id": "provenance_required", + "kind": "policy_check", + "description": "A configured provenance requirement was not satisfied for the component.", + "finding_buckets": [] + }, + "scorecard_below_threshold": { + "rule_id": "scorecard_below_threshold", + "kind": "policy_check", + "description": "A mapped repository's OpenSSF Scorecard score was below the configured minimum threshold.", + "finding_buckets": [] } }, + "provenance_summary": { + "components_in_scope": 2, + "pypi_components_in_scope": 2, + "pypi_components_without_provenance": 2, + "components_with_provenance": 0, + "components_with_attestations": 0, + "components_with_attestation_gaps": 0, + "components_with_enrichment_errors": 0, + "unsupported_components": 0 + }, + "attestation_summary": { + "files_evaluated": 0, + "files_with_attestations": 0, + "files_without_attestations": 0, + "packages_with_attestation_gaps": [], + "publisher_kind_counts": {} + }, + "scorecard_summary": { + "enabled": false, + "components_in_scope": 2, + "candidate_components": 0, + "supported_components": 0, + "components_with_mapped_repositories": 0, + "components_with_scorecards": 0, + "scorecard_unavailable": 0, + "repository_unmapped": 0, + "components_with_enrichment_errors": 0, + "results": [] + }, + "enrichment_metadata": { + "mode": "offline_default", + "pypi_enabled": false, + "pypi_timeout_seconds": null, + "pypi_network_access_performed": false, + "network_access_performed": false, + "candidate_components": 0, + "supported_components": 0, + "status_counts": {}, + "scorecard_enabled": false, + "scorecard_timeout_seconds": null, + "scorecard_network_access_performed": false, + "scorecard_candidate_components": 0, + "scorecard_supported_components": 0, + "scorecard_status_counts": {} + }, + "trust_signal_notes": [ + "PyPI components are present, but provenance enrichment was not enabled for this run." + ], "metadata": { "before_format": "cyclonedx-json", "after_format": "cyclonedx-json", @@ -401,6 +479,22 @@ "ignored_checks": 0 }, "exit_code": 0 + }, + "enrichment": { + "mode": "offline_default", + "pypi_enabled": false, + "pypi_timeout_seconds": null, + "pypi_network_access_performed": false, + "network_access_performed": false, + "candidate_components": 0, + "supported_components": 0, + "status_counts": {}, + "scorecard_enabled": false, + "scorecard_timeout_seconds": null, + "scorecard_network_access_performed": false, + "scorecard_candidate_components": 0, + "scorecard_supported_components": 0, + "scorecard_status_counts": {} } }, "notes": [ diff --git a/tools/sbom-diff-and-risk/examples/sample-report.md b/tools/sbom-diff-and-risk/examples/sample-report.md index 6569589..5b73e63 100644 --- a/tools/sbom-diff-and-risk/examples/sample-report.md +++ b/tools/sbom-diff-and-risk/examples/sample-report.md @@ -24,6 +24,56 @@ - Warnings: 0 - Suppressed findings: 0 +## Provenance summary +- Enrichment mode: offline_default +- Network access performed: no +- Candidate components for enrichment: 0 +- Supported components for enrichment: 0 +- Observed provenance status counts: none +- Components in scope: 2 +- PyPI components in scope: 2 +- PyPI components without provenance records: 2 +- Components with provenance evidence: 0 +- Components with attestations: 0 +- Components with attestation gaps: 0 +- Components with enrichment errors: 0 +- Unsupported components: 0 + +## Attestation gaps +| component | version | statuses | +|-----------|---------|----------| +| _none_ | | | + +## Policy impact for provenance-related rules +| rule id | component | level | message | +|---------|-----------|-------|---------| +| _none_ | | | | + +## Trust signal notes +- PyPI components are present, but provenance enrichment was not enabled for this run. + +## Scorecard summary +- Enrichment enabled: no +- Network access performed: no +- Candidate components for Scorecard enrichment: 0 +- Components with supported repository mappings: 0 +- Components with mapped repositories: 0 +- Components with available Scorecards: 0 +- Scorecard unavailable: 0 +- Repository unmapped: 0 +- Components with enrichment errors: 0 +- Observed Scorecard status counts: none + +## Scorecard results +| component | version | repository | score | status | +|-----------|---------|------------|-------|--------| +| _none_ | | | | | + +## Policy impact for Scorecard-related rules +| rule id | component | level | message | +|---------|-----------|-------|---------| +| _none_ | | | | + ## Added components | name | version | ecosystem | risk buckets | |------|---------|-----------|--------------| diff --git a/tools/sbom-diff-and-risk/examples/sample-requirements-report.json b/tools/sbom-diff-and-risk/examples/sample-requirements-report.json index 4d8427e..2cda00e 100644 --- a/tools/sbom-diff-and-risk/examples/sample-requirements-report.json +++ b/tools/sbom-diff-and-risk/examples/sample-requirements-report.json @@ -315,8 +315,86 @@ "kind": "policy_check", "description": "Component source host was not present in the configured allow_sources list.", "finding_buckets": [] + }, + "missing_attestation": { + "rule_id": "missing_attestation", + "kind": "provenance_signal", + "description": "PyPI release metadata was fetched, but no attestations were published for the package release.", + "finding_buckets": [] + }, + "unverified_provenance": { + "rule_id": "unverified_provenance", + "kind": "provenance_signal", + "description": "PyPI attestations were present, but provenance could not be verified against publisher metadata.", + "finding_buckets": [] + }, + "provenance_unavailable": { + "rule_id": "provenance_unavailable", + "kind": "provenance_signal", + "description": "PyPI provenance evidence was unavailable because enrichment was disabled, unsupported, or errored.", + "finding_buckets": [] + }, + "provenance_required": { + "rule_id": "provenance_required", + "kind": "policy_check", + "description": "A configured provenance requirement was not satisfied for the component.", + "finding_buckets": [] + }, + "scorecard_below_threshold": { + "rule_id": "scorecard_below_threshold", + "kind": "policy_check", + "description": "A mapped repository's OpenSSF Scorecard score was below the configured minimum threshold.", + "finding_buckets": [] } }, + "provenance_summary": { + "components_in_scope": 2, + "pypi_components_in_scope": 2, + "pypi_components_without_provenance": 2, + "components_with_provenance": 0, + "components_with_attestations": 0, + "components_with_attestation_gaps": 0, + "components_with_enrichment_errors": 0, + "unsupported_components": 0 + }, + "attestation_summary": { + "files_evaluated": 0, + "files_with_attestations": 0, + "files_without_attestations": 0, + "packages_with_attestation_gaps": [], + "publisher_kind_counts": {} + }, + "scorecard_summary": { + "enabled": false, + "components_in_scope": 2, + "candidate_components": 0, + "supported_components": 0, + "components_with_mapped_repositories": 0, + "components_with_scorecards": 0, + "scorecard_unavailable": 0, + "repository_unmapped": 0, + "components_with_enrichment_errors": 0, + "results": [] + }, + "enrichment_metadata": { + "mode": "offline_default", + "pypi_enabled": false, + "pypi_timeout_seconds": null, + "pypi_network_access_performed": false, + "network_access_performed": false, + "candidate_components": 0, + "supported_components": 0, + "status_counts": {}, + "scorecard_enabled": false, + "scorecard_timeout_seconds": null, + "scorecard_network_access_performed": false, + "scorecard_candidate_components": 0, + "scorecard_supported_components": 0, + "scorecard_status_counts": {} + }, + "trust_signal_notes": [ + "PyPI components are present, but provenance enrichment was not enabled for this run." + ], "metadata": { "before_format": "requirements-txt", "after_format": "requirements-txt", @@ -337,6 +415,22 @@ "ignored_checks": 0 }, "exit_code": 0 + }, + "enrichment": { + "mode": "offline_default", + "pypi_enabled": false, + "pypi_timeout_seconds": null, + "pypi_network_access_performed": false, + "network_access_performed": false, + "candidate_components": 0, + "supported_components": 0, + "status_counts": {}, + "scorecard_enabled": false, + "scorecard_timeout_seconds": null, + "scorecard_network_access_performed": false, + "scorecard_candidate_components": 0, + "scorecard_supported_components": 0, + "scorecard_status_counts": {} } }, "notes": [ diff --git a/tools/sbom-diff-and-risk/examples/sample-requirements-report.md b/tools/sbom-diff-and-risk/examples/sample-requirements-report.md index 9564d50..5969247 100644 --- a/tools/sbom-diff-and-risk/examples/sample-requirements-report.md +++ b/tools/sbom-diff-and-risk/examples/sample-requirements-report.md @@ -24,6 +24,56 @@ - Warnings: 0 - Suppressed findings: 0 +## Provenance summary +- Enrichment mode: offline_default +- Network access performed: no +- Candidate components for enrichment: 0 +- Supported components for enrichment: 0 +- Observed provenance status counts: none +- Components in scope: 2 +- PyPI components in scope: 2 +- PyPI components without provenance records: 2 +- Components with provenance evidence: 0 +- Components with attestations: 0 +- Components with attestation gaps: 0 +- Components with enrichment errors: 0 +- Unsupported components: 0 + +## Attestation gaps +| component | version | statuses | +|-----------|---------|----------| +| _none_ | | | + +## Policy impact for provenance-related rules +| rule id | component | level | message | +|---------|-----------|-------|---------| +| _none_ | | | | + +## Trust signal notes +- PyPI components are present, but provenance enrichment was not enabled for this run. + +## Scorecard summary +- Enrichment enabled: no +- Network access performed: no +- Candidate components for Scorecard enrichment: 0 +- Components with supported repository mappings: 0 +- Components with mapped repositories: 0 +- Components with available Scorecards: 0 +- Scorecard unavailable: 0 +- Repository unmapped: 0 +- Components with enrichment errors: 0 +- Observed Scorecard status counts: none + +## Scorecard results +| component | version | repository | score | status | +|-----------|---------|------------|-------|--------| +| _none_ | | | | | + +## Policy impact for Scorecard-related rules +| rule id | component | level | message | +|---------|-----------|-------|---------| +| _none_ | | | | + ## Added components | name | version | ecosystem | risk buckets | |------|---------|-----------|--------------| diff --git a/tools/sbom-diff-and-risk/examples/sample-sarif.sarif b/tools/sbom-diff-and-risk/examples/sample-sarif.sarif index 7ddd5e5..2f18a5b 100644 --- a/tools/sbom-diff-and-risk/examples/sample-sarif.sarif +++ b/tools/sbom-diff-and-risk/examples/sample-sarif.sarif @@ -7,8 +7,8 @@ "driver": { "name": "sbom-diff-risk", "fullName": "sbom-diff-risk", - "version": "0.2.0", - "semanticVersion": "0.2.0", + "version": "0.3.0", + "semanticVersion": "0.3.0", "rules": [ { "id": "sdr.major_upgrade", @@ -33,7 +33,7 @@ "id": "sdr.policy_violation.allow_sources", "name": "policy_violation.allow_sources", "shortDescription": { - "text": "Blocking policy violation: allow_sources" + "text": "Policy violation: allow_sources" }, "fullDescription": { "text": "Component source host was not present in the configured allow_sources list." @@ -52,7 +52,7 @@ "id": "sdr.policy_violation.max_added_packages", "name": "policy_violation.max_added_packages", "shortDescription": { - "text": "Blocking policy violation: max_added_packages" + "text": "Policy violation: max_added_packages" }, "fullDescription": { "text": "Added package count exceeded the configured deterministic threshold." @@ -292,7 +292,7 @@ ], "originalUriBaseIds": { "%SRCROOT%": { - "uri": "file:///D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/" + "uri": "file:///__PROJECT_ROOT__/" } } } diff --git a/tools/sbom-diff-and-risk/examples/sample-scorecard-report.json b/tools/sbom-diff-and-risk/examples/sample-scorecard-report.json new file mode 100644 index 0000000..211b91d --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/sample-scorecard-report.json @@ -0,0 +1,567 @@ +{ + "summary": { + "added": 2, + "removed": 0, + "changed": 1, + "risk_counts": { + "new_package": 2, + "major_upgrade": 0, + "version_change_unclassified": 0, + "unknown_license": 0, + "stale_package": 0, + "suspicious_source": 0, + "not_evaluated": 0 + } + }, + "components": { + "added": [ + { + "name": "requests", + "version": "2.32.0", + "ecosystem": "pypi", + "purl": "pkg:pypi/requests@2.32.0", + "license_id": null, + "supplier": null, + "source_url": "https://github.com/psf/requests", + "bom_ref": null, + "raw_type": null, + "evidence": { + "scorecard": { + "provider": "openssf-scorecard", + "requested": true, + "repository": { + "platform": "github.com", + "owner": "psf", + "repo": "requests", + "canonical_name": "github.com/psf/requests", + "repository_url": "https://github.com/psf/requests", + "source": "component.source_url", + "confidence": "high" + }, + "statuses": [ + "scorecard_available" + ], + "score": 6.0, + "date": "2026-04-10T00:00:00Z", + "scorecard_version": "5.0.0", + "scorecard_commit": "def456", + "repository_commit": "abc123", + "checks": [ + { + "name": "Maintained", + "score": 10, + "reason": "Project is active.", + "documentation_url": null, + "documentation_short": null + }, + { + "name": "Binary-Artifacts", + "score": 8, + "reason": "No unexpected artifacts were found.", + "documentation_url": null, + "documentation_short": null + } + ], + "note": null, + "error": null + } + } + }, + { + "name": "urllib3", + "version": "2.2.1", + "ecosystem": "pypi", + "purl": "pkg:pypi/urllib3@2.2.1", + "license_id": null, + "supplier": null, + "source_url": "https://pypi.org/project/urllib3/2.2.1/", + "bom_ref": null, + "raw_type": null, + "evidence": { + "scorecard": { + "provider": "openssf-scorecard", + "requested": true, + "repository": null, + "statuses": [ + "repository_unmapped" + ], + "score": null, + "date": null, + "scorecard_version": null, + "scorecard_commit": null, + "repository_commit": null, + "checks": [], + "note": "No high-confidence source repository mapping was available from explicit component metadata.", + "error": null + } + } + } + ], + "removed": [], + "changed": [ + { + "key": "purl:pkg:pypi/certifi", + "classification": "version_changed", + "before": { + "name": "certifi", + "version": "2025.1.0", + "ecosystem": "pypi", + "purl": "pkg:pypi/certifi@2025.1.0", + "license_id": null, + "supplier": null, + "source_url": null, + "bom_ref": null, + "raw_type": null, + "evidence": {} + }, + "after": { + "name": "certifi", + "version": "2026.1.1", + "ecosystem": "pypi", + "purl": "pkg:pypi/certifi@2026.1.1", + "license_id": null, + "supplier": null, + "source_url": "https://github.com/certifi/python-certifi", + "bom_ref": null, + "raw_type": null, + "evidence": { + "scorecard": { + "provider": "openssf-scorecard", + "requested": true, + "repository": { + "platform": "github.com", + "owner": "certifi", + "repo": "python-certifi", + "canonical_name": "github.com/certifi/python-certifi", + "repository_url": "https://github.com/certifi/python-certifi", + "source": "component.source_url", + "confidence": "high" + }, + "statuses": [ + "scorecard_available" + ], + "score": 8.4, + "date": "2026-04-10T00:00:00Z", + "scorecard_version": "5.0.0", + "scorecard_commit": "def456", + "repository_commit": "abc123", + "checks": [ + { + "name": "Maintained", + "score": 10, + "reason": "Project is active.", + "documentation_url": null, + "documentation_short": null + }, + { + "name": "Binary-Artifacts", + "score": 8, + "reason": "No unexpected artifacts were found.", + "documentation_url": null, + "documentation_short": null + } + ], + "note": null, + "error": null + } + } + } + } + ] + }, + "risks": [ + { + "bucket": "new_package", + "component_key": "purl:pkg:pypi/requests", + "component": { + "name": "requests", + "version": "2.32.0", + "ecosystem": "pypi", + "purl": "pkg:pypi/requests@2.32.0", + "license_id": null, + "supplier": null, + "source_url": "https://github.com/psf/requests", + "bom_ref": null, + "raw_type": null, + "evidence": { + "scorecard": { + "provider": "openssf-scorecard", + "requested": true, + "repository": { + "platform": "github.com", + "owner": "psf", + "repo": "requests", + "canonical_name": "github.com/psf/requests", + "repository_url": "https://github.com/psf/requests", + "source": "component.source_url", + "confidence": "high" + }, + "statuses": [ + "scorecard_available" + ], + "score": 6.0, + "date": "2026-04-10T00:00:00Z", + "scorecard_version": "5.0.0", + "scorecard_commit": "def456", + "repository_commit": "abc123", + "checks": [ + { + "name": "Maintained", + "score": 10, + "reason": "Project is active.", + "documentation_url": null, + "documentation_short": null + }, + { + "name": "Binary-Artifacts", + "score": 8, + "reason": "No unexpected artifacts were found.", + "documentation_url": null, + "documentation_short": null + } + ], + "note": null, + "error": null + } + } + }, + "rationale": "Component was not present in the before input." + }, + { + "bucket": "new_package", + "component_key": "purl:pkg:pypi/urllib3", + "component": { + "name": "urllib3", + "version": "2.2.1", + "ecosystem": "pypi", + "purl": "pkg:pypi/urllib3@2.2.1", + "license_id": null, + "supplier": null, + "source_url": "https://pypi.org/project/urllib3/2.2.1/", + "bom_ref": null, + "raw_type": null, + "evidence": { + "scorecard": { + "provider": "openssf-scorecard", + "requested": true, + "repository": null, + "statuses": [ + "repository_unmapped" + ], + "score": null, + "date": null, + "scorecard_version": null, + "scorecard_commit": null, + "repository_commit": null, + "checks": [], + "note": "No high-confidence source repository mapping was available from explicit component metadata.", + "error": null + } + } + }, + "rationale": "Component was not present in the before input." + } + ], + "policy_evaluation": { + "applied": true, + "policy_path": "examples/policy-scorecard-minimal.yml", + "effective_policy": { + "version": 3, + "block_on": [], + "warn_on": [ + "scorecard_below_threshold" + ], + "max_added_packages": null, + "allow_sources": [], + "ignore_rules": [], + "require_attestations_for_new_packages": false, + "require_provenance_for_suspicious_sources": false, + "allow_unattested_packages": [], + "allow_provenance_publishers": [], + "minimum_scorecard_score": 7.0 + }, + "blocking_violations": [], + "warning_violations": [ + { + "rule_id": "scorecard_below_threshold", + "level": "warn", + "message": "Scorecard score 6.0 is below minimum_scorecard_score=7.0 for repository github.com/psf/requests.", + "component_key": "purl:pkg:pypi/requests", + "component_name": "requests", + "finding_bucket": null, + "suppression_reason": null + } + ], + "suppressed_violations": [], + "totals": { + "blocking": 0, + "warning": 1, + "suppressed": 0, + "ignored_checks": 0 + }, + "exit_code": 0 + }, + "blocking_findings": [], + "warning_findings": [ + { + "rule_id": "scorecard_below_threshold", + "level": "warn", + "message": "Scorecard score 6.0 is below minimum_scorecard_score=7.0 for repository github.com/psf/requests.", + "component_key": "purl:pkg:pypi/requests", + "component_name": "requests", + "finding_bucket": null, + "suppression_reason": null + } + ], + "suppressed_findings": [], + "rule_catalog": { + "new_package": { + "rule_id": "new_package", + "kind": "risk_finding", + "description": "Component is present only in the after input.", + "finding_buckets": [ + "new_package" + ] + }, + "major_upgrade": { + "rule_id": "major_upgrade", + "kind": "risk_finding", + "description": "Version change is a parseable SemVer major upgrade.", + "finding_buckets": [ + "major_upgrade" + ] + }, + "version_change_unclassified": { + "rule_id": "version_change_unclassified", + "kind": "risk_finding", + "description": "Version changed but could not be classified as a reliable major SemVer upgrade.", + "finding_buckets": [ + "version_change_unclassified" + ] + }, + "unknown_license": { + "rule_id": "unknown_license", + "kind": "risk_finding", + "description": "License metadata is missing, empty, UNKNOWN, or NOASSERTION.", + "finding_buckets": [ + "unknown_license" + ] + }, + "suspicious_source": { + "rule_id": "suspicious_source", + "kind": "risk_finding", + "description": "Source provenance is missing or points to a suspicious scheme, path, or host.", + "finding_buckets": [ + "suspicious_source" + ] + }, + "stale_package": { + "rule_id": "stale_package", + "kind": "risk_finding", + "description": "Staleness check result. Offline mode maps this rule to not_evaluated instead of guessing.", + "finding_buckets": [ + "stale_package", + "not_evaluated" + ] + }, + "max_added_packages": { + "rule_id": "max_added_packages", + "kind": "policy_check", + "description": "Added package count exceeded the configured deterministic threshold.", + "finding_buckets": [] + }, + "allow_sources": { + "rule_id": "allow_sources", + "kind": "policy_check", + "description": "Component source host was not present in the configured allow_sources list.", + "finding_buckets": [] + }, + "missing_attestation": { + "rule_id": "missing_attestation", + "kind": "provenance_signal", + "description": "PyPI release metadata was fetched, but no attestations were published for the package release.", + "finding_buckets": [] + }, + "unverified_provenance": { + "rule_id": "unverified_provenance", + "kind": "provenance_signal", + "description": "PyPI attestations were present, but provenance could not be verified against publisher metadata.", + "finding_buckets": [] + }, + "provenance_unavailable": { + "rule_id": "provenance_unavailable", + "kind": "provenance_signal", + "description": "PyPI provenance evidence was unavailable because enrichment was disabled, unsupported, or errored.", + "finding_buckets": [] + }, + "provenance_required": { + "rule_id": "provenance_required", + "kind": "policy_check", + "description": "A configured provenance requirement was not satisfied for the component.", + "finding_buckets": [] + }, + "scorecard_below_threshold": { + "rule_id": "scorecard_below_threshold", + "kind": "policy_check", + "description": "A mapped repository's OpenSSF Scorecard score was below the configured minimum threshold.", + "finding_buckets": [] + } + }, + "provenance_summary": { + "components_in_scope": 3, + "pypi_components_in_scope": 3, + "pypi_components_without_provenance": 3, + "components_with_provenance": 0, + "components_with_attestations": 0, + "components_with_attestation_gaps": 0, + "components_with_enrichment_errors": 0, + "unsupported_components": 0 + }, + "attestation_summary": { + "files_evaluated": 0, + "files_with_attestations": 0, + "files_without_attestations": 0, + "packages_with_attestation_gaps": [], + "publisher_kind_counts": {} + }, + "scorecard_summary": { + "enabled": true, + "components_in_scope": 3, + "candidate_components": 3, + "supported_components": 2, + "components_with_mapped_repositories": 2, + "components_with_scorecards": 2, + "scorecard_unavailable": 0, + "repository_unmapped": 1, + "components_with_enrichment_errors": 0, + "results": [ + { + "component_key": "purl:pkg:pypi/requests", + "name": "requests", + "version": "2.32.0", + "repository": "github.com/psf/requests", + "repository_source": "component.source_url", + "status": "scorecard_available", + "score": 6.0, + "note": null, + "error": null + }, + { + "component_key": "purl:pkg:pypi/urllib3", + "name": "urllib3", + "version": "2.2.1", + "repository": null, + "repository_source": null, + "status": "repository_unmapped", + "score": null, + "note": "No high-confidence source repository mapping was available from explicit component metadata.", + "error": null + }, + { + "component_key": "purl:pkg:pypi/certifi", + "name": "certifi", + "version": "2026.1.1", + "repository": "github.com/certifi/python-certifi", + "repository_source": "component.source_url", + "status": "scorecard_available", + "score": 8.4, + "note": null, + "error": null + } + ] + }, + "enrichment_metadata": { + "mode": "opt_in_scorecard", + "pypi_enabled": false, + "pypi_timeout_seconds": null, + "pypi_network_access_performed": false, + "network_access_performed": true, + "candidate_components": 0, + "supported_components": 0, + "status_counts": {}, + "scorecard_enabled": true, + "scorecard_timeout_seconds": 3.0, + "scorecard_network_access_performed": true, + "scorecard_candidate_components": 3, + "scorecard_supported_components": 2, + "scorecard_status_counts": { + "repository_unmapped": 1, + "scorecard_available": 2 + } + }, + "trust_signal_notes": [ + "PyPI components are present, but provenance enrichment was not enabled for this run.", + "OpenSSF Scorecard results are auxiliary trust signals and are not proof of safety.", + "Scorecard lookups are skipped when no high-confidence repository mapping is available.", + "Policy produced 1 Scorecard-related blocking or warning decision(s)." + ], + "metadata": { + "before_format": "requirements-txt", + "after_format": "requirements-txt", + "generated_at": null, + "strict": false, + "stub": false, + "policy_evaluation": { + "applied": true, + "policy_path": "examples/policy-scorecard-minimal.yml", + "effective_policy": { + "version": 3, + "block_on": [], + "warn_on": [ + "scorecard_below_threshold" + ], + "max_added_packages": null, + "allow_sources": [], + "ignore_rules": [], + "require_attestations_for_new_packages": false, + "require_provenance_for_suspicious_sources": false, + "allow_unattested_packages": [], + "allow_provenance_publishers": [], + "minimum_scorecard_score": 7.0 + }, + "blocking_violations": [], + "warning_violations": [ + { + "rule_id": "scorecard_below_threshold", + "level": "warn", + "message": "Scorecard score 6.0 is below minimum_scorecard_score=7.0 for repository github.com/psf/requests.", + "component_key": "purl:pkg:pypi/requests", + "component_name": "requests", + "finding_bucket": null, + "suppression_reason": null + } + ], + "suppressed_violations": [], + "totals": { + "blocking": 0, + "warning": 1, + "suppressed": 0, + "ignored_checks": 0 + }, + "exit_code": 0 + }, + "enrichment": { + "mode": "opt_in_scorecard", + "pypi_enabled": false, + "pypi_timeout_seconds": null, + "pypi_network_access_performed": false, + "network_access_performed": true, + "candidate_components": 0, + "supported_components": 0, + "status_counts": {}, + "scorecard_enabled": true, + "scorecard_timeout_seconds": 3.0, + "scorecard_network_access_performed": true, + "scorecard_candidate_components": 3, + "scorecard_supported_components": 2, + "scorecard_status_counts": { + "repository_unmapped": 1, + "scorecard_available": 2 + } + } + }, + "notes": [ + "This tool uses heuristic risk classification.", + "OpenSSF Scorecard enrichment was requested explicitly." + ] +} diff --git a/tools/sbom-diff-and-risk/examples/sample-scorecard-report.md b/tools/sbom-diff-and-risk/examples/sample-scorecard-report.md new file mode 100644 index 0000000..cf24ba2 --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/sample-scorecard-report.md @@ -0,0 +1,116 @@ +# sbom-diff-and-risk report + +## Summary +- Before format: requirements-txt +- After format: requirements-txt +- Added: 2 +- Removed: 0 +- Version changes: 1 + +## Risk buckets +- new_package: 2 +- major_upgrade: 0 +- version_change_unclassified: 0 +- unknown_license: 0 +- stale_package: 0 +- suspicious_source: 0 +- not_evaluated: 0 + +## Policy summary +- Applied: yes +- Policy path: examples/policy-scorecard-minimal.yml +- Exit code: 0 +- Blocking findings: 0 +- Warnings: 1 +- Suppressed findings: 0 + +## Provenance summary +- Enrichment mode: opt_in_scorecard +- Network access performed: no +- Candidate components for enrichment: 0 +- Supported components for enrichment: 0 +- Observed provenance status counts: none +- Components in scope: 3 +- PyPI components in scope: 3 +- PyPI components without provenance records: 3 +- Components with provenance evidence: 0 +- Components with attestations: 0 +- Components with attestation gaps: 0 +- Components with enrichment errors: 0 +- Unsupported components: 0 + +## Attestation gaps +| component | version | statuses | +|-----------|---------|----------| +| _none_ | | | + +## Policy impact for provenance-related rules +| rule id | component | level | message | +|---------|-----------|-------|---------| +| _none_ | | | | + +## Trust signal notes +- PyPI components are present, but provenance enrichment was not enabled for this run. +- OpenSSF Scorecard results are auxiliary trust signals and are not proof of safety. +- Scorecard lookups are skipped when no high-confidence repository mapping is available. +- Policy produced 1 Scorecard-related blocking or warning decision(s). + +## Scorecard summary +- Enrichment enabled: yes +- Network access performed: yes +- Candidate components for Scorecard enrichment: 3 +- Components with supported repository mappings: 2 +- Components with mapped repositories: 2 +- Components with available Scorecards: 2 +- Scorecard unavailable: 0 +- Repository unmapped: 1 +- Components with enrichment errors: 0 +- Observed Scorecard status counts: repository_unmapped=1, scorecard_available=2 + +## Scorecard results +| component | version | repository | score | status | +|-----------|---------|------------|-------|--------| +| requests | 2.32.0 | github.com/psf/requests | 6.0 | scorecard_available | +| urllib3 | 2.2.1 | | | repository_unmapped | +| certifi | 2026.1.1 | github.com/certifi/python-certifi | 8.4 | scorecard_available | + +## Policy impact for Scorecard-related rules +| rule id | component | level | message | +|---------|-----------|-------|---------| +| scorecard_below_threshold | requests | warn | Scorecard score 6.0 is below minimum_scorecard_score=7.0 for repository github.com/psf/requests. | + +## Added components +| name | version | ecosystem | risk buckets | +|------|---------|-----------|--------------| +| requests | 2.32.0 | pypi | new_package | +| urllib3 | 2.2.1 | pypi | new_package | + +## Removed components +| name | version | ecosystem | +|------|---------|-----------| +| _none_ | | | + +## Version changes +| name | before | after | classification | risk buckets | +|------|--------|-------|----------------|--------------| +| certifi | 2025.1.0 | 2026.1.1 | version_changed | | + +## Risk findings +| bucket | component | version | rationale | +|--------|-----------|---------|-----------| +| new_package | requests | 2.32.0 | Component was not present in the before input. | +| new_package | urllib3 | 2.2.1 | Component was not present in the before input. | + +## Blocking violations +| rule id | component | level | message | +|---------|-----------|-------|---------| +| _none_ | | | | + +## Warnings +| rule id | component | level | message | +|---------|-----------|-------|---------| +| scorecard_below_threshold | requests | warn | Scorecard score 6.0 is below minimum_scorecard_score=7.0 for repository github.com/psf/requests. | + +## Notes +- This tool uses heuristic risk classification. +- OpenSSF Scorecard enrichment was requested explicitly. diff --git a/tools/sbom-diff-and-risk/examples/sample-scorecard-report.sarif b/tools/sbom-diff-and-risk/examples/sample-scorecard-report.sarif new file mode 100644 index 0000000..25048a4 --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/sample-scorecard-report.sarif @@ -0,0 +1,100 @@ +{ + "$schema": "https://json.schemastore.org/sarif-2.1.0.json", + "version": "2.1.0", + "runs": [ + { + "tool": { + "driver": { + "name": "sbom-diff-risk", + "fullName": "sbom-diff-risk", + "version": "0.3.0", + "semanticVersion": "0.3.0", + "rules": [ + { + "id": "sdr.policy_violation.scorecard_below_threshold", + "name": "policy_violation.scorecard_below_threshold", + "shortDescription": { + "text": "Policy violation: scorecard_below_threshold" + }, + "fullDescription": { + "text": "A mapped repository's OpenSSF Scorecard score was below the configured minimum threshold." + }, + "defaultConfiguration": { + "level": "warning" + }, + "properties": { + "tags": [ + "supply-chain", + "policy", + "scorecard" + ] + } + } + ] + } + }, + "artifacts": [ + { + "location": { + "uri": "examples/requirements_before.txt", + "uriBaseId": "%SRCROOT%" + } + }, + { + "location": { + "uri": "examples/requirements_after.txt", + "uriBaseId": "%SRCROOT%" + } + } + ], + "properties": { + "sbom_diff_risk": { + "result_limit": 5000, + "total_candidate_results": 1, + "emitted_results": 1, + "omitted_results": 0, + "truncated": false, + "prioritization": "error results first, then warning, then note; direct mapped findings before policy-only checks; stable rule priority and component key tie-breakers.", + "warning": null + } + }, + "results": [ + { + "ruleId": "sdr.policy_violation.scorecard_below_threshold", + "level": "warning", + "message": { + "text": "requests: Scorecard score 6.0 is below minimum_scorecard_score=7.0 for repository github.com/psf/requests." + }, + "locations": [ + { + "physicalLocation": { + "artifactLocation": { + "uri": "examples/requirements_after.txt", + "uriBaseId": "%SRCROOT%" + }, + "region": { + "startLine": 1 + } + } + } + ], + "partialFingerprints": { + "ruleId": "sdr.policy_violation.scorecard_below_threshold", + "componentKey": "purl:pkg:pypi/requests" + }, + "properties": { + "policy_rule_id": "scorecard_below_threshold", + "component_key": "purl:pkg:pypi/requests", + "component_name": "requests", + "result_kind": "policy_violation" + } + } + ], + "originalUriBaseIds": { + "%SRCROOT%": { + "uri": "file:///__PROJECT_ROOT__/" + } + } + } + ] +} diff --git a/tools/sbom-diff-and-risk/pyproject.toml b/tools/sbom-diff-and-risk/pyproject.toml index ddc6b3c..2664052 100644 --- a/tools/sbom-diff-and-risk/pyproject.toml +++ b/tools/sbom-diff-and-risk/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "sbom-diff-and-risk" -version = "0.2.0" +version = "0.3.0" description = "Local, deterministic SBOM diff and heuristic risk reporting." readme = "README.md" requires-python = ">=3.11" diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/__init__.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/__init__.py index 0236f6e..f9f24ce 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/__init__.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/__init__.py @@ -2,4 +2,4 @@ __all__ = ["__version__"] -__version__ = "0.2.0" +__version__ = "0.3.0" diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/cli.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/cli.py index 8faf99f..007d372 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/cli.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/cli.py @@ -6,8 +6,9 @@ from typing import Sequence from .diffing import diff_components +from .enrichment import DEFAULT_PYPI_TIMEOUT_SECONDS, PyPIProvenanceEnricher, merge_enrichment_metadata from .errors import ParseError, PolicyError -from .models import CompareReport, ReportComponents, ReportMetadata, ReportSummary +from .models import CompareReport, ReportComponents, ReportEnrichmentMetadata, ReportMetadata, ReportSummary from .normalize import SUPPORTED_FORMATS, normalize_input_with_options from .policy_evaluator import evaluate_policy from .policy_parser import build_policy @@ -16,6 +17,7 @@ from .report_md import render_report_markdown from .report_sarif import render_report_sarif_output from .risk import evaluate_risks, summarize_risks +from .scorecard_enrichment import DEFAULT_SCORECARD_TIMEOUT_SECONDS, ScorecardEnricher def build_parser() -> argparse.ArgumentParser: @@ -68,7 +70,7 @@ def build_parser() -> argparse.ArgumentParser: "--policy", type=Path, default=None, - help="Apply a YAML policy v1 file. Blocking violations return exit code 1.", + help="Apply a YAML policy v1, v2, or v3 file. Blocking violations return exit code 1.", ) compare.add_argument( "--fail-on", @@ -88,7 +90,24 @@ def build_parser() -> argparse.ArgumentParser: compare.add_argument( "--enrich-pypi", action="store_true", - help="Reserved for future PyPI metadata enrichment. Not implemented in v0.1 scaffolding.", + help="Opt-in PyPI provenance and integrity enrichment. Default behavior remains offline with no network access.", + ) + compare.add_argument( + "--pypi-timeout", + type=float, + default=DEFAULT_PYPI_TIMEOUT_SECONDS, + help="Timeout in seconds for each explicit PyPI enrichment request. Used only with --enrich-pypi.", + ) + compare.add_argument( + "--enrich-scorecard", + action="store_true", + help="Opt-in OpenSSF Scorecard enrichment for components with high-confidence repository mappings.", + ) + compare.add_argument( + "--scorecard-timeout", + type=float, + default=DEFAULT_SCORECARD_TIMEOUT_SECONDS, + help="Timeout in seconds for each explicit Scorecard enrichment request. Used only with --enrich-scorecard.", ) compare.add_argument( "--source-allowlist", @@ -113,12 +132,17 @@ def main(argv: Sequence[str] | None = None) -> int: def run_compare(args: argparse.Namespace) -> int: pyproject_group = getattr(args, "pyproject_group", None) - - if args.enrich_pypi: - raise NotImplementedError("--enrich-pypi is reserved for a later network-enabled release.") + enrich_pypi = getattr(args, "enrich_pypi", False) + enrich_scorecard = getattr(args, "enrich_scorecard", False) + pypi_timeout = getattr(args, "pypi_timeout", DEFAULT_PYPI_TIMEOUT_SECONDS) + scorecard_timeout = getattr(args, "scorecard_timeout", DEFAULT_SCORECARD_TIMEOUT_SECONDS) if args.out_json is None and args.out_md is None and args.out_sarif is None: raise ValueError("at least one of --out-json, --out-md, or --out-sarif must be provided") + if pypi_timeout <= 0: + raise ValueError("--pypi-timeout must be a positive number of seconds.") + if scorecard_timeout <= 0: + raise ValueError("--scorecard-timeout must be a positive number of seconds.") before_path: Path = args.before after_path: Path = args.after @@ -141,6 +165,23 @@ def run_compare(args: argparse.Namespace) -> int: ) if pyproject_group and before_format != "pyproject-toml" and after_format != "pyproject-toml": raise ValueError("--pyproject-group requires at least one pyproject.toml input.") + + pypi_enrichment_metadata = None + if enrich_pypi: + enricher = PyPIProvenanceEnricher(timeout_seconds=pypi_timeout) + before_components = enricher.enrich_components(before_components) + after_components = enricher.enrich_components(after_components) + pypi_enrichment_metadata = enricher.build_report_metadata() + + scorecard_enrichment_metadata = None + if enrich_scorecard: + scorecard_enricher = ScorecardEnricher(timeout_seconds=scorecard_timeout) + before_components = scorecard_enricher.enrich_components(before_components) + after_components = scorecard_enricher.enrich_components(after_components) + scorecard_enrichment_metadata = scorecard_enricher.build_report_metadata() + + enrichment_metadata = merge_enrichment_metadata(pypi_enrichment_metadata, scorecard_enrichment_metadata) + policy, policy_path = build_policy(policy_path=args.policy, fail_on=args.fail_on, warn_on=args.warn_on) added, removed, changed = diff_components(before_components, after_components) @@ -156,7 +197,7 @@ def run_compare(args: argparse.Namespace) -> int: notes = [ "This tool uses heuristic risk classification.", - "No network enrichment was performed.", + _enrichment_note(enrich_pypi, enrich_scorecard), *before_notes, *after_notes, ] @@ -177,6 +218,7 @@ def run_compare(args: argparse.Namespace) -> int: strict=args.strict, stub=False, policy_evaluation=policy_evaluation, + enrichment=enrichment_metadata or ReportEnrichmentMetadata(), ), notes=notes, ) @@ -246,5 +288,15 @@ def _format_policy_failure_summary(policy_evaluation) -> str: return "\n".join(lines) +def _enrichment_note(enrich_pypi: bool, enrich_scorecard: bool) -> str: + if enrich_pypi and enrich_scorecard: + return "PyPI provenance and OpenSSF Scorecard enrichment were requested explicitly." + if enrich_pypi: + return "PyPI provenance enrichment was requested explicitly." + if enrich_scorecard: + return "OpenSSF Scorecard enrichment was requested explicitly." + return "No network enrichment was performed." + + if __name__ == "__main__": raise SystemExit(main()) diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/enrichment.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/enrichment.py new file mode 100644 index 0000000..ea7f951 --- /dev/null +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/enrichment.py @@ -0,0 +1,149 @@ +from __future__ import annotations + +from collections import Counter +from dataclasses import replace + +from .models import ( + Component, + ProvenanceEvidence, + ReportEnrichmentMetadata, +) +from .pypi_integrity_client import PyPIIntegrityClient +from .pypi_provenance import ( + normalize_provenance_file, + normalize_pypi_provenance, + provenance_evidence_to_dict, +) + +DEFAULT_PYPI_TIMEOUT_SECONDS = 5.0 + + +class PyPIProvenanceEnricher: + def __init__( + self, + *, + client: PyPIIntegrityClient | None = None, + timeout_seconds: float = DEFAULT_PYPI_TIMEOUT_SECONDS, + ) -> None: + self.client = client or PyPIIntegrityClient(timeout_seconds=timeout_seconds) + self.timeout_seconds = timeout_seconds + self._cache: dict[tuple[str, str, str], ProvenanceEvidence] = {} + self._seen_keys: set[tuple[str, str, str]] = set() + + def enrich_components(self, components: list[Component]) -> list[Component]: + enriched: list[Component] = [] + for component in components: + key = _component_identity(component) + if key not in self._cache: + self._seen_keys.add(key) + self._cache[key] = normalize_pypi_provenance(component, client=self.client) + enriched.append(replace(component, provenance=self._cache[key])) + return enriched + + def build_report_metadata(self) -> ReportEnrichmentMetadata: + if not self._seen_keys: + return ReportEnrichmentMetadata( + mode="opt_in_pypi", + pypi_enabled=True, + pypi_timeout_seconds=self.timeout_seconds, + pypi_network_access_performed=False, + network_access_performed=False, + candidate_components=0, + supported_components=0, + status_counts={}, + ) + + evidences = [self._cache[key] for key in sorted(self._seen_keys)] + counter = Counter( + status.value + for evidence in evidences + for status in evidence.statuses + ) + pypi_network_access_performed = any(evidence.lookup_performed for evidence in evidences) + return ReportEnrichmentMetadata( + mode="opt_in_pypi", + pypi_enabled=True, + pypi_timeout_seconds=self.timeout_seconds, + pypi_network_access_performed=pypi_network_access_performed, + network_access_performed=pypi_network_access_performed, + candidate_components=len(evidences), + supported_components=sum(1 for evidence in evidences if evidence.supported), + status_counts={key: counter[key] for key in sorted(counter)}, + ) + + +def merge_enrichment_metadata(*metadata_items: ReportEnrichmentMetadata | None) -> ReportEnrichmentMetadata: + items = [item for item in metadata_items if item is not None] + if not items: + return ReportEnrichmentMetadata() + + merged = ReportEnrichmentMetadata( + mode=_combined_enrichment_mode(items), + pypi_enabled=any(item.pypi_enabled for item in items), + pypi_timeout_seconds=_first_non_none(item.pypi_timeout_seconds for item in items), + pypi_network_access_performed=any(item.pypi_network_access_performed for item in items), + scorecard_enabled=any(item.scorecard_enabled for item in items), + scorecard_timeout_seconds=_first_non_none(item.scorecard_timeout_seconds for item in items), + scorecard_network_access_performed=any(item.scorecard_network_access_performed for item in items), + network_access_performed=any(item.network_access_performed for item in items), + candidate_components=sum(item.candidate_components for item in items), + supported_components=sum(item.supported_components for item in items), + status_counts=_merge_status_counts(item.status_counts for item in items), + scorecard_candidate_components=sum(item.scorecard_candidate_components for item in items), + scorecard_supported_components=sum(item.scorecard_supported_components for item in items), + scorecard_status_counts=_merge_status_counts(item.scorecard_status_counts for item in items), + ) + return merged + + +def enrichment_metadata_to_dict(metadata: ReportEnrichmentMetadata) -> dict[str, object]: + return { + "mode": metadata.mode, + "pypi_enabled": metadata.pypi_enabled, + "pypi_timeout_seconds": metadata.pypi_timeout_seconds, + "pypi_network_access_performed": metadata.pypi_network_access_performed, + "network_access_performed": metadata.network_access_performed, + "candidate_components": metadata.candidate_components, + "supported_components": metadata.supported_components, + "status_counts": dict(metadata.status_counts), + "scorecard_enabled": metadata.scorecard_enabled, + "scorecard_timeout_seconds": metadata.scorecard_timeout_seconds, + "scorecard_network_access_performed": metadata.scorecard_network_access_performed, + "scorecard_candidate_components": metadata.scorecard_candidate_components, + "scorecard_supported_components": metadata.scorecard_supported_components, + "scorecard_status_counts": dict(metadata.scorecard_status_counts), + } + + +def _component_identity(component: Component) -> tuple[str, str, str]: + return ( + component.ecosystem.strip().lower(), + component.name.strip().lower(), + (component.version or "").strip().lower(), + ) + + +def _combined_enrichment_mode(metadata_items: list[ReportEnrichmentMetadata]) -> str: + pypi_enabled = any(item.pypi_enabled for item in metadata_items) + scorecard_enabled = any(item.scorecard_enabled for item in metadata_items) + if pypi_enabled and scorecard_enabled: + return "opt_in_pypi_and_scorecard" + if pypi_enabled: + return "opt_in_pypi" + if scorecard_enabled: + return "opt_in_scorecard" + return "offline_default" + + +def _merge_status_counts(status_count_dicts) -> dict[str, int]: # noqa: ANN001 + counter: Counter[str] = Counter() + for status_counts in status_count_dicts: + counter.update(status_counts) + return {key: counter[key] for key in sorted(counter)} + + +def _first_non_none(values) -> float | None: # noqa: ANN001 + for value in values: + if value is not None: + return value + return None diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/models.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/models.py index ca97cf6..9c399c6 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/models.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/models.py @@ -18,6 +18,111 @@ class RiskBucket(StrEnum): NOT_EVALUATED = "not_evaluated" +class ProvenanceStatus(StrEnum): + PROVENANCE_AVAILABLE = "provenance_available" + ATTESTATION_AVAILABLE = "attestation_available" + ATTESTATION_UNAVAILABLE = "attestation_unavailable" + ENRICHMENT_ERROR = "enrichment_error" + UNSUPPORTED_FOR_PACKAGE = "unsupported_for_package" + + +class RepositoryMappingConfidence(StrEnum): + HIGH = "high" + LOW = "low" + + +@dataclass(slots=True, frozen=True) +class RepositoryMapping: + platform: str + owner: str + repo: str + canonical_name: str + repository_url: str + source: str + confidence: RepositoryMappingConfidence = RepositoryMappingConfidence.HIGH + + +class ScorecardStatus(StrEnum): + SCORECARD_AVAILABLE = "scorecard_available" + SCORECARD_UNAVAILABLE = "scorecard_unavailable" + REPOSITORY_UNMAPPED = "repository_unmapped" + ENRICHMENT_ERROR = "enrichment_error" + + +@dataclass(slots=True, frozen=True) +class ProvenanceFileEvidence: + filename: str + url: str | None = None + sha256: str | None = None + upload_time: str | None = None + yanked: bool | None = None + statuses: tuple[ProvenanceStatus, ...] = () + attestation_count: int = 0 + predicate_types: tuple[str, ...] = () + publisher_kinds: tuple[str, ...] = () + error: str | None = None + + +@dataclass(slots=True, frozen=True) +class ProvenanceEvidence: + provider: str + requested: bool + supported: bool = False + lookup_performed: bool = False + package_name: str | None = None + package_version: str | None = None + release_url: str | None = None + statuses: tuple[ProvenanceStatus, ...] = () + files: tuple[ProvenanceFileEvidence, ...] = () + files_evaluated: int = 0 + files_with_attestations: int = 0 + files_without_attestations: int = 0 + error: str | None = None + + +@dataclass(slots=True, frozen=True) +class ScorecardCheck: + name: str + score: int + reason: str | None = None + documentation_url: str | None = None + documentation_short: str | None = None + + +@dataclass(slots=True, frozen=True) +class ScorecardEvidence: + provider: str + requested: bool + repository: RepositoryMapping | None = None + statuses: tuple[ScorecardStatus, ...] = () + score: float | None = None + date: str | None = None + scorecard_version: str | None = None + scorecard_commit: str | None = None + repository_commit: str | None = None + checks: tuple[ScorecardCheck, ...] = () + note: str | None = None + error: str | None = None + + +@dataclass(slots=True) +class ReportEnrichmentMetadata: + mode: str = "offline_default" + pypi_enabled: bool = False + pypi_timeout_seconds: float | None = None + pypi_network_access_performed: bool = False + network_access_performed: bool = False + candidate_components: int = 0 + supported_components: int = 0 + status_counts: dict[str, int] = field(default_factory=dict) + scorecard_enabled: bool = False + scorecard_timeout_seconds: float | None = None + scorecard_network_access_performed: bool = False + scorecard_candidate_components: int = 0 + scorecard_supported_components: int = 0 + scorecard_status_counts: dict[str, int] = field(default_factory=dict) + + @dataclass(slots=True) class Component: name: str @@ -30,6 +135,8 @@ class Component: bom_ref: str | None = None raw_type: str | None = None evidence: dict[str, Any] = field(default_factory=dict) + provenance: ProvenanceEvidence | None = None + scorecard: ScorecardEvidence | None = None @dataclass(slots=True) @@ -71,6 +178,7 @@ class ReportMetadata: strict: bool = False stub: bool = True policy_evaluation: PolicyEvaluation | None = None + enrichment: ReportEnrichmentMetadata = field(default_factory=ReportEnrichmentMetadata) @dataclass(slots=True) diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_evaluator.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_evaluator.py index c52ff0d..72d329a 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_evaluator.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_evaluator.py @@ -1,12 +1,23 @@ from __future__ import annotations +from dataclasses import dataclass from urllib.parse import urlparse from .diffing import component_key -from .models import Component, ComponentChange, RiskBucket, RiskFinding +from .models import Component, ComponentChange, ProvenanceStatus, RiskBucket, RiskFinding, ScorecardStatus from .policy_models import PolicyConfig, PolicyEvaluation, PolicyLevel, PolicyViolation +@dataclass(slots=True, frozen=True) +class ProvenanceAssessment: + attestation_available: bool + provenance_unavailable: bool + verified: bool + publisher_kinds: tuple[str, ...] + unavailable_message: str | None = None + unverified_message: str | None = None + + def evaluate_policy( policy: PolicyConfig | None, *, @@ -26,71 +37,191 @@ def evaluate_policy( for finding in findings: rule_id = finding_rule_id(finding) severity = _severity_for_rule(policy, rule_id) - if rule_id in policy.ignore_rules: - ignored_checks += 1 - suppressed_violations.append( - PolicyViolation( - rule_id=rule_id, + ignored_checks += _record_violation( + policy, + severity=severity, + violation=PolicyViolation( + rule_id=rule_id, + level=severity, + message=finding.rationale, + component_key=finding.component_key, + component_name=finding.component.name, + finding_bucket=finding.bucket.value, + ), + blocking_violations=blocking_violations, + warning_violations=warning_violations, + suppressed_violations=suppressed_violations, + ) + + provenance_components = _components_for_policy(added, changed) + added_pypi_keys = { + component_key(item) + for item in added + if item.ecosystem.strip().lower() == "pypi" + } + suspicious_source_keys = { + finding.component_key + for finding in findings + if finding_rule_id(finding) == "suspicious_source" + } + for component in provenance_components: + if component.ecosystem.strip().lower() != "pypi": + continue + package_is_unattested_allowed = _package_is_unattested_allowed(policy, component) + assessment = _assess_provenance(component, policy) + + # Keep allow_unattested_packages narrow and explicit: it waives only + # missing-attestation checks, not complete provenance unavailability. + if assessment.provenance_unavailable: + severity = _severity_for_rule(policy, "provenance_unavailable") + ignored_checks += _record_violation( + policy, + severity=severity, + violation=PolicyViolation( + rule_id="provenance_unavailable", level=severity, - message=finding.rationale, - component_key=finding.component_key, - component_name=finding.component.name, - finding_bucket=finding.bucket.value, - suppression_reason="ignored_by_policy", - ) + message=assessment.unavailable_message or "PyPI provenance evidence is unavailable.", + component_key=component_key(component), + component_name=component.name, + ), + blocking_violations=blocking_violations, + warning_violations=warning_violations, + suppressed_violations=suppressed_violations, + ) + elif not assessment.attestation_available and not package_is_unattested_allowed: + severity = _severity_for_rule(policy, "missing_attestation") + ignored_checks += _record_violation( + policy, + severity=severity, + violation=PolicyViolation( + rule_id="missing_attestation", + level=severity, + message="PyPI release metadata was fetched, but no attestations were published for this package release.", + component_key=component_key(component), + component_name=component.name, + ), + blocking_violations=blocking_violations, + warning_violations=warning_violations, + suppressed_violations=suppressed_violations, ) - continue - if severity is None: - continue + if assessment.attestation_available and not assessment.verified: + severity = _severity_for_rule(policy, "unverified_provenance") + ignored_checks += _record_violation( + policy, + severity=severity, + violation=PolicyViolation( + rule_id="unverified_provenance", + level=severity, + message=assessment.unverified_message or "PyPI provenance could not be verified.", + component_key=component_key(component), + component_name=component.name, + ), + blocking_violations=blocking_violations, + warning_violations=warning_violations, + suppressed_violations=suppressed_violations, + ) - violation = PolicyViolation( - rule_id=rule_id, - level=severity, - message=finding.rationale, - component_key=finding.component_key, - component_name=finding.component.name, - finding_bucket=finding.bucket.value, - ) - _append_violation(violation, blocking_violations, warning_violations) + requirement_contexts: list[str] = [] + if policy.require_attestations_for_new_packages and component_key(component) in added_pypi_keys: + requirement_contexts.append("new package") + if ( + policy.require_provenance_for_suspicious_sources + and component_key(component) in suspicious_source_keys + ): + requirement_contexts.append("suspicious source") + + if requirement_contexts: + requirement_message = _provenance_requirement_message( + assessment, + requirement_contexts, + package_is_unattested_allowed=package_is_unattested_allowed, + ) + if requirement_message is not None: + severity = _severity_for_rule(policy, "provenance_required", default=PolicyLevel.BLOCK) + ignored_checks += _record_violation( + policy, + severity=severity, + violation=PolicyViolation( + rule_id="provenance_required", + level=severity, + message=requirement_message, + component_key=component_key(component), + component_name=component.name, + ), + blocking_violations=blocking_violations, + warning_violations=warning_violations, + suppressed_violations=suppressed_violations, + ) if policy.max_added_packages is not None and len(added) > policy.max_added_packages: rule_id = "max_added_packages" severity = _severity_for_rule(policy, rule_id, default=PolicyLevel.BLOCK) - violation = PolicyViolation( - rule_id=rule_id, - level=severity, - message=f"Added package count {len(added)} exceeds max_added_packages={policy.max_added_packages}.", - suppression_reason="ignored_by_policy" if rule_id in policy.ignore_rules else None, + ignored_checks += _record_violation( + policy, + severity=severity, + violation=PolicyViolation( + rule_id=rule_id, + level=severity, + message=f"Added package count {len(added)} exceeds max_added_packages={policy.max_added_packages}.", + ), + blocking_violations=blocking_violations, + warning_violations=warning_violations, + suppressed_violations=suppressed_violations, ) - if rule_id in policy.ignore_rules: - ignored_checks += 1 - suppressed_violations.append(violation) - elif severity is not None: - _append_violation(violation, blocking_violations, warning_violations) if policy.allow_sources: - for component in _components_for_source_policy(added, changed): + for component in provenance_components: host = _source_host(component.source_url) if host in policy.allow_sources: continue rule_id = "allow_sources" severity = _severity_for_rule(policy, rule_id, default=PolicyLevel.BLOCK) - violation = PolicyViolation( - rule_id=rule_id, - level=severity, - message=f"Source host {host or 'missing'} is not present in allow_sources.", - component_key=component_key(component), - component_name=component.name, - suppression_reason="ignored_by_policy" if rule_id in policy.ignore_rules else None, - ) - if rule_id in policy.ignore_rules: - ignored_checks += 1 - suppressed_violations.append(violation) + ignored_checks += _record_violation( + policy, + severity=severity, + violation=PolicyViolation( + rule_id=rule_id, + level=severity, + message=f"Source host {host or 'missing'} is not present in allow_sources.", + component_key=component_key(component), + component_name=component.name, + ), + blocking_violations=blocking_violations, + warning_violations=warning_violations, + suppressed_violations=suppressed_violations, + ) + + if policy.minimum_scorecard_score is not None: + rule_id = "scorecard_below_threshold" + severity = _severity_for_rule(policy, rule_id) + for component in provenance_components: + scorecard_score = _scorecard_score(component) + if scorecard_score is None or scorecard_score >= policy.minimum_scorecard_score: continue - if severity is not None: - _append_violation(violation, blocking_violations, warning_violations) + repository_name = ( + component.scorecard.repository.canonical_name + if component.scorecard is not None and component.scorecard.repository is not None + else "unmapped-repository" + ) + ignored_checks += _record_violation( + policy, + severity=severity, + violation=PolicyViolation( + rule_id=rule_id, + level=severity, + message=( + f"Scorecard score {scorecard_score:.1f} is below minimum_scorecard_score=" + f"{policy.minimum_scorecard_score:.1f} for repository {repository_name}." + ), + component_key=component_key(component), + component_name=component.name, + ), + blocking_violations=blocking_violations, + warning_violations=warning_violations, + suppressed_violations=suppressed_violations, + ) blocking_violations.sort(key=_violation_sort_key) warning_violations.sort(key=_violation_sort_key) @@ -128,6 +259,46 @@ def _severity_for_rule( return default +def _record_violation( + policy: PolicyConfig, + *, + severity: PolicyLevel | None, + violation: PolicyViolation, + blocking_violations: list[PolicyViolation], + warning_violations: list[PolicyViolation], + suppressed_violations: list[PolicyViolation], +) -> int: + if violation.rule_id in policy.ignore_rules: + suppressed_violations.append( + PolicyViolation( + rule_id=violation.rule_id, + level=severity, + message=violation.message, + component_key=violation.component_key, + component_name=violation.component_name, + finding_bucket=violation.finding_bucket, + suppression_reason="ignored_by_policy", + ) + ) + return 1 + if severity is None: + return 0 + + _append_violation( + PolicyViolation( + rule_id=violation.rule_id, + level=severity, + message=violation.message, + component_key=violation.component_key, + component_name=violation.component_name, + finding_bucket=violation.finding_bucket, + ), + blocking_violations, + warning_violations, + ) + return 0 + + def _append_violation( violation: PolicyViolation, blocking_violations: list[PolicyViolation], @@ -139,12 +310,122 @@ def _append_violation( warning_violations.append(violation) -def _components_for_source_policy(added: list[Component], changed: list[ComponentChange]) -> list[Component]: +def _components_for_policy(added: list[Component], changed: list[ComponentChange]) -> list[Component]: components = list(added) components.extend(change.after for change in changed) return components +def _package_is_unattested_allowed(policy: PolicyConfig, component: Component) -> bool: + return component.name.strip().lower() in set(policy.allow_unattested_packages) + + +def _assess_provenance(component: Component, policy: PolicyConfig) -> ProvenanceAssessment: + provenance = component.provenance + if provenance is None: + return ProvenanceAssessment( + attestation_available=False, + provenance_unavailable=True, + verified=False, + publisher_kinds=(), + unavailable_message="PyPI provenance evidence is unavailable because enrichment was not enabled for this run.", + ) + + status_set = set(provenance.statuses) + if ProvenanceStatus.ENRICHMENT_ERROR in status_set: + return ProvenanceAssessment( + attestation_available=False, + provenance_unavailable=True, + verified=False, + publisher_kinds=(), + unavailable_message=provenance.error or "PyPI provenance evidence could not be fetched due to an enrichment error.", + ) + if ProvenanceStatus.UNSUPPORTED_FOR_PACKAGE in status_set: + return ProvenanceAssessment( + attestation_available=False, + provenance_unavailable=True, + verified=False, + publisher_kinds=(), + unavailable_message="PyPI provenance evidence is unavailable for this package or version.", + ) + + attestation_available = _status_present(provenance.statuses, ProvenanceStatus.ATTESTATION_AVAILABLE) or any( + _status_present(item.statuses, ProvenanceStatus.ATTESTATION_AVAILABLE) + for item in provenance.files + ) + publisher_kinds = tuple( + sorted( + { + publisher.strip().lower() + for item in provenance.files + for publisher in item.publisher_kinds + if publisher.strip() + } + ) + ) + verified = _publishers_are_verified(publisher_kinds, policy.allow_provenance_publishers) if attestation_available else False + unverified_message = None + if attestation_available and not verified: + if policy.allow_provenance_publishers: + allowlist = ", ".join(policy.allow_provenance_publishers) + actual = ", ".join(publisher_kinds) if publisher_kinds else "missing" + unverified_message = ( + f"PyPI attestations were present, but publisher kinds {actual} did not match " + f"allow_provenance_publishers={allowlist}." + ) + else: + unverified_message = "PyPI attestations were present, but no publisher identity was available to verify provenance." + + return ProvenanceAssessment( + attestation_available=attestation_available, + provenance_unavailable=False, + verified=verified, + publisher_kinds=publisher_kinds, + unverified_message=unverified_message, + ) + + +def _status_present(statuses: tuple[ProvenanceStatus, ...], candidate: ProvenanceStatus) -> bool: + return candidate in statuses + + +def _publishers_are_verified(publisher_kinds: tuple[str, ...], allowlist: tuple[str, ...]) -> bool: + if not publisher_kinds: + return False + if not allowlist: + return True + return any(publisher in set(allowlist) for publisher in publisher_kinds) + + +def _scorecard_score(component: Component) -> float | None: + if component.scorecard is None: + return None + if ScorecardStatus.SCORECARD_AVAILABLE not in component.scorecard.statuses: + return None + return component.scorecard.score + + +def _provenance_requirement_message( + assessment: ProvenanceAssessment, + requirement_contexts: list[str], + *, + package_is_unattested_allowed: bool, +) -> str | None: + context_label = " and ".join(requirement_contexts) + if assessment.provenance_unavailable: + return f"Provenance is required for {context_label}, but evidence is unavailable: {assessment.unavailable_message}" + if not assessment.attestation_available: + if package_is_unattested_allowed: + return None + return f"Provenance is required for {context_label}, but no attestations were published for this PyPI package." + if not assessment.verified: + return ( + f"Provenance is required for {context_label}, but the available attestations could not be verified: " + f"{assessment.unverified_message}" + ) + return None + + def _source_host(source_url: str | None) -> str | None: if not source_url: return None diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_models.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_models.py index ad2362f..4b1105a 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_models.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_models.py @@ -9,7 +9,7 @@ class PolicyLevel(StrEnum): WARN = "warn" -SUPPORTED_POLICY_RULE_IDS = ( +V1_SUPPORTED_POLICY_RULE_IDS = ( "new_package", "major_upgrade", "version_change_unclassified", @@ -20,6 +20,23 @@ class PolicyLevel(StrEnum): "allow_sources", ) +V2_PROVENANCE_POLICY_RULE_IDS = ( + "missing_attestation", + "unverified_provenance", + "provenance_unavailable", + "provenance_required", +) + +V3_SCORECARD_POLICY_RULE_IDS = ( + "scorecard_below_threshold", +) + +SUPPORTED_POLICY_RULE_IDS = ( + *V1_SUPPORTED_POLICY_RULE_IDS, + *V2_PROVENANCE_POLICY_RULE_IDS, + *V3_SCORECARD_POLICY_RULE_IDS, +) + @dataclass(slots=True, frozen=True) class PolicyConfig: @@ -29,6 +46,11 @@ class PolicyConfig: max_added_packages: int | None = None allow_sources: tuple[str, ...] = () ignore_rules: tuple[str, ...] = () + require_attestations_for_new_packages: bool = False + require_provenance_for_suspicious_sources: bool = False + allow_unattested_packages: tuple[str, ...] = () + allow_provenance_publishers: tuple[str, ...] = () + minimum_scorecard_score: float | None = None @dataclass(slots=True) diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_parser.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_parser.py index eb11b5c..189d7fa 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_parser.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_parser.py @@ -6,9 +6,15 @@ import yaml from .errors import PolicyError -from .policy_models import PolicyConfig, SUPPORTED_POLICY_RULE_IDS - -_SUPPORTED_POLICY_KEYS = { +from .policy_models import ( + PolicyConfig, + SUPPORTED_POLICY_RULE_IDS, + V1_SUPPORTED_POLICY_RULE_IDS, + V2_PROVENANCE_POLICY_RULE_IDS, + V3_SCORECARD_POLICY_RULE_IDS, +) + +_V1_SUPPORTED_POLICY_KEYS = { "version", "block_on", "warn_on", @@ -17,6 +23,20 @@ "ignore_rules", } +_V2_ONLY_POLICY_KEYS = { + "require_attestations_for_new_packages", + "require_provenance_for_suspicious_sources", + "allow_unattested_packages", + "allow_provenance_publishers", + "allow_unattested_publishers", +} + +_V3_ONLY_POLICY_KEYS = { + "minimum_scorecard_score", +} + +_SUPPORTED_POLICY_KEYS = _V1_SUPPORTED_POLICY_KEYS | _V2_ONLY_POLICY_KEYS | _V3_ONLY_POLICY_KEYS + def load_policy(path: Path) -> PolicyConfig: if not path.is_file(): @@ -37,18 +57,76 @@ def load_policy(path: Path) -> PolicyConfig: version = payload.get("version") if not isinstance(version, int): raise PolicyError(f"Invalid policy schema in {path}: version must be an integer.") - if version != 1: - raise PolicyError(f"Invalid policy schema in {path}: only version 1 is supported.") + if version not in {1, 2, 3}: + raise PolicyError(f"Invalid policy schema in {path}: only versions 1, 2, and 3 are supported.") + + if version == 1: + version_unknown_keys = sorted(set(payload) & (_V2_ONLY_POLICY_KEYS | _V3_ONLY_POLICY_KEYS)) + if version_unknown_keys: + raise PolicyError( + f"Invalid policy schema in {path}: version 1 does not support keys: {', '.join(version_unknown_keys)}." + ) + if version == 2: + version_unknown_keys = sorted(set(payload) & _V3_ONLY_POLICY_KEYS) + if version_unknown_keys: + raise PolicyError( + f"Invalid policy schema in {path}: version 2 does not support keys: {', '.join(version_unknown_keys)}." + ) + + if "allow_provenance_publishers" in payload and "allow_unattested_publishers" in payload: + raise PolicyError( + f"Invalid policy schema in {path}: use either allow_provenance_publishers or " + "allow_unattested_publishers, not both." + ) - block_on = _parse_rule_list(payload.get("block_on", []), f"{path}: block_on") - warn_on = _parse_rule_list(payload.get("warn_on", []), f"{path}: warn_on") - ignore_rules = _parse_rule_list(payload.get("ignore_rules", []), f"{path}: ignore_rules") + if version == 1: + supported_rule_ids = V1_SUPPORTED_POLICY_RULE_IDS + elif version == 2: + supported_rule_ids = (*V1_SUPPORTED_POLICY_RULE_IDS, *V2_PROVENANCE_POLICY_RULE_IDS) + else: + supported_rule_ids = SUPPORTED_POLICY_RULE_IDS + + block_on = _parse_rule_list(payload.get("block_on", []), f"{path}: block_on", supported_rule_ids=supported_rule_ids) + warn_on = _parse_rule_list(payload.get("warn_on", []), f"{path}: warn_on", supported_rule_ids=supported_rule_ids) + ignore_rules = _parse_rule_list( + payload.get("ignore_rules", []), + f"{path}: ignore_rules", + supported_rule_ids=supported_rule_ids, + ) max_added_packages = payload.get("max_added_packages") if max_added_packages is not None and (not isinstance(max_added_packages, int) or max_added_packages < 0): raise PolicyError(f"Invalid policy schema in {path}: max_added_packages must be a non-negative integer.") allow_sources = _parse_string_list(payload.get("allow_sources", []), f"{path}: allow_sources", lower=True) + require_attestations_for_new_packages = _parse_bool( + payload.get("require_attestations_for_new_packages", False), + f"{path}: require_attestations_for_new_packages", + ) + require_provenance_for_suspicious_sources = _parse_bool( + payload.get("require_provenance_for_suspicious_sources", False), + f"{path}: require_provenance_for_suspicious_sources", + ) + allow_unattested_packages = _parse_string_list( + payload.get("allow_unattested_packages", []), + f"{path}: allow_unattested_packages", + lower=True, + ) + allow_provenance_publishers_value = payload.get("allow_provenance_publishers") + if allow_provenance_publishers_value is None: + allow_provenance_publishers_value = payload.get("allow_unattested_publishers", []) + allow_provenance_publishers_context = f"{path}: allow_unattested_publishers" + else: + allow_provenance_publishers_context = f"{path}: allow_provenance_publishers" + allow_provenance_publishers = _parse_string_list( + allow_provenance_publishers_value, + allow_provenance_publishers_context, + lower=True, + ) + minimum_scorecard_score = _parse_optional_score( + payload.get("minimum_scorecard_score"), + f"{path}: minimum_scorecard_score", + ) return normalize_policy( PolicyConfig( @@ -58,6 +136,11 @@ def load_policy(path: Path) -> PolicyConfig: max_added_packages=max_added_packages, allow_sources=allow_sources, ignore_rules=ignore_rules, + require_attestations_for_new_packages=require_attestations_for_new_packages, + require_provenance_for_suspicious_sources=require_provenance_for_suspicious_sources, + allow_unattested_packages=allow_unattested_packages, + allow_provenance_publishers=allow_provenance_publishers, + minimum_scorecard_score=minimum_scorecard_score, ) ) @@ -87,21 +170,32 @@ def build_policy( max_added_packages=seed.max_added_packages, allow_sources=seed.allow_sources, ignore_rules=seed.ignore_rules, + require_attestations_for_new_packages=seed.require_attestations_for_new_packages, + require_provenance_for_suspicious_sources=seed.require_provenance_for_suspicious_sources, + allow_unattested_packages=seed.allow_unattested_packages, + allow_provenance_publishers=seed.allow_provenance_publishers, + minimum_scorecard_score=seed.minimum_scorecard_score, ) return normalize_policy(merged), rendered_path def normalize_policy(policy: PolicyConfig) -> PolicyConfig: + normalized_version = max(policy.version, _required_policy_version(policy)) block_on = tuple(dict.fromkeys(policy.block_on)) warn_on = tuple(rule for rule in dict.fromkeys(policy.warn_on) if rule not in block_on) ignore_rules = tuple(dict.fromkeys(policy.ignore_rules)) return PolicyConfig( - version=policy.version, + version=normalized_version, block_on=block_on, warn_on=warn_on, max_added_packages=policy.max_added_packages, allow_sources=tuple(dict.fromkeys(policy.allow_sources)), ignore_rules=ignore_rules, + require_attestations_for_new_packages=policy.require_attestations_for_new_packages, + require_provenance_for_suspicious_sources=policy.require_provenance_for_suspicious_sources, + allow_unattested_packages=tuple(dict.fromkeys(policy.allow_unattested_packages)), + allow_provenance_publishers=tuple(dict.fromkeys(policy.allow_provenance_publishers)), + minimum_scorecard_score=policy.minimum_scorecard_score, ) @@ -112,17 +206,17 @@ def parse_rule_csv(value: str | None, source_name: str) -> tuple[str, ...]: parsed = [entry for entry in entries if entry] if not parsed: raise PolicyError(f"{source_name} requires at least one rule id.") - return _validate_rule_ids(parsed, source_name) + return _validate_rule_ids(parsed, source_name, supported_rule_ids=SUPPORTED_POLICY_RULE_IDS) -def _parse_rule_list(value: object, context: str) -> tuple[str, ...]: +def _parse_rule_list(value: object, context: str, *, supported_rule_ids: Iterable[str]) -> tuple[str, ...]: if value is None: return () if not isinstance(value, list): raise PolicyError(f"Invalid policy schema in {context}: expected a YAML list of rule ids.") if not all(isinstance(item, str) for item in value): raise PolicyError(f"Invalid policy schema in {context}: all rule ids must be strings.") - return _validate_rule_ids(value, context) + return _validate_rule_ids(value, context, supported_rule_ids=supported_rule_ids) def _parse_string_list(value: object, context: str, *, lower: bool = False) -> tuple[str, ...]: @@ -141,13 +235,31 @@ def _parse_string_list(value: object, context: str, *, lower: bool = False) -> t return tuple(dict.fromkeys(items)) -def _validate_rule_ids(rule_ids: Iterable[str], context: str) -> tuple[str, ...]: +def _parse_bool(value: object, context: str) -> bool: + if not isinstance(value, bool): + raise PolicyError(f"Invalid policy schema in {context}: expected a boolean value.") + return value + + +def _parse_optional_score(value: object, context: str) -> float | None: + if value is None: + return None + if not isinstance(value, (int, float)): + raise PolicyError(f"Invalid policy schema in {context}: expected a number between 0 and 10.") + normalized = float(value) + if normalized < 0 or normalized > 10: + raise PolicyError(f"Invalid policy schema in {context}: expected a number between 0 and 10.") + return normalized + + +def _validate_rule_ids(rule_ids: Iterable[str], context: str, *, supported_rule_ids: Iterable[str]) -> tuple[str, ...]: normalized = tuple(dict.fromkeys(rule_id.strip() for rule_id in rule_ids if rule_id.strip())) - unknown = sorted(set(normalized) - set(SUPPORTED_POLICY_RULE_IDS)) + supported = set(supported_rule_ids) + unknown = sorted(set(normalized) - supported) if unknown: raise PolicyError( f"Unknown rule id(s) in {context}: {', '.join(unknown)}. " - f"Supported rule ids: {', '.join(SUPPORTED_POLICY_RULE_IDS)}." + f"Supported rule ids: {', '.join(sorted(supported))}." ) return normalized @@ -156,6 +268,24 @@ def _merge_strings(base: tuple[str, ...], extra: tuple[str, ...]) -> tuple[str, return tuple(dict.fromkeys((*base, *extra))) +def _required_policy_version(policy: PolicyConfig) -> int: + if any(rule in V3_SCORECARD_POLICY_RULE_IDS for rule in (*policy.block_on, *policy.warn_on, *policy.ignore_rules)): + return 3 + if policy.minimum_scorecard_score is not None: + return 3 + if any(rule in V2_PROVENANCE_POLICY_RULE_IDS for rule in (*policy.block_on, *policy.warn_on, *policy.ignore_rules)): + return 2 + if policy.require_attestations_for_new_packages: + return 2 + if policy.require_provenance_for_suspicious_sources: + return 2 + if policy.allow_unattested_packages: + return 2 + if policy.allow_provenance_publishers: + return 2 + return 1 + + def _render_policy_path(policy_path: Path) -> str: resolved_policy_path = policy_path.resolve() try: diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/presentation.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/presentation.py index 421f661..d41fcb7 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/presentation.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/presentation.py @@ -1,9 +1,19 @@ from __future__ import annotations +from collections import Counter from dataclasses import dataclass from typing import Any -from .policy_models import PolicyConfig, PolicyEvaluation, PolicyViolation +from .diffing import component_key +from .enrichment import enrichment_metadata_to_dict +from .models import CompareReport, Component, ProvenanceStatus, ScorecardStatus +from .policy_models import ( + PolicyConfig, + PolicyEvaluation, + PolicyViolation, + V2_PROVENANCE_POLICY_RULE_IDS, + V3_SCORECARD_POLICY_RULE_IDS, +) @dataclass(slots=True, frozen=True) @@ -61,13 +71,41 @@ class RuleCatalogEntry: kind="policy_check", description="Component source host was not present in the configured allow_sources list.", ), + RuleCatalogEntry( + rule_id="missing_attestation", + kind="provenance_signal", + description="PyPI release metadata was fetched, but no attestations were published for the package release.", + ), + RuleCatalogEntry( + rule_id="unverified_provenance", + kind="provenance_signal", + description="PyPI attestations were present, but provenance could not be verified against publisher metadata.", + ), + RuleCatalogEntry( + rule_id="provenance_unavailable", + kind="provenance_signal", + description="PyPI provenance evidence was unavailable because enrichment was disabled, unsupported, or errored.", + ), + RuleCatalogEntry( + rule_id="provenance_required", + kind="policy_check", + description="A configured provenance requirement was not satisfied for the component.", + ), + RuleCatalogEntry( + rule_id="scorecard_below_threshold", + kind="policy_check", + description="A mapped repository's OpenSSF Scorecard score was below the configured minimum threshold.", + ), ) def build_policy_report_sections(policy_evaluation: PolicyEvaluation | None) -> dict[str, Any]: evaluation_dict = policy_evaluation_to_dict(policy_evaluation) + provenance_policy = provenance_policy_summary(policy_evaluation) return { "policy_evaluation": evaluation_dict, + "provenance_policy": provenance_policy, + "provenance_policy_impact": provenance_policy, "blocking_findings": [ policy_violation_to_dict(item) for item in effective_policy_evaluation(policy_evaluation).blocking_violations ], @@ -109,7 +147,7 @@ def policy_evaluation_to_dict(policy_evaluation: PolicyEvaluation | None) -> dic def policy_config_to_dict(policy: PolicyConfig | None) -> dict[str, Any] | None: if policy is None: return None - return { + payload = { "version": policy.version, "block_on": list(policy.block_on), "warn_on": list(policy.warn_on), @@ -117,6 +155,22 @@ def policy_config_to_dict(policy: PolicyConfig | None) -> dict[str, Any] | None: "allow_sources": list(policy.allow_sources), "ignore_rules": list(policy.ignore_rules), } + if policy.version >= 2: + payload.update( + { + "require_attestations_for_new_packages": policy.require_attestations_for_new_packages, + "require_provenance_for_suspicious_sources": policy.require_provenance_for_suspicious_sources, + "allow_unattested_packages": list(policy.allow_unattested_packages), + "allow_provenance_publishers": list(policy.allow_provenance_publishers), + } + ) + if policy.version >= 3: + payload.update( + { + "minimum_scorecard_score": policy.minimum_scorecard_score, + } + ) + return payload def policy_violation_to_dict(violation: PolicyViolation) -> dict[str, Any]: @@ -148,3 +202,257 @@ def summarize_violations_by_rule(violations: list[PolicyViolation]) -> list[tupl for violation in violations: counts[violation.rule_id] = counts.get(violation.rule_id, 0) + 1 return sorted(counts.items()) + + +def build_trust_signal_report_sections(report: CompareReport) -> dict[str, Any]: + components = _current_state_components(report) + pypi_components = [component for component in components if component.ecosystem.strip().lower() == "pypi"] + provenance_components = [component for component in components if component.provenance is not None] + scorecard_components = [component for component in components if component.scorecard is not None] + publisher_counts = Counter( + publisher + for component in provenance_components + for file_evidence in component.provenance.files + for publisher in file_evidence.publisher_kinds + ) + packages_with_attestation_gaps = [ + { + "component_key": component_key(component), + "name": component.name, + "version": component.version, + "statuses": [status.value for status in component.provenance.statuses], + } + for component in provenance_components + if _component_has_attestation_gap(component) + ] + provenance_summary = { + "components_in_scope": len(components), + "pypi_components_in_scope": len(pypi_components), + "pypi_components_without_provenance": sum(1 for component in pypi_components if component.provenance is None), + "components_with_provenance": sum(1 for component in provenance_components if _component_has_status(component, ProvenanceStatus.PROVENANCE_AVAILABLE)), + "components_with_attestations": sum(1 for component in provenance_components if _component_has_status(component, ProvenanceStatus.ATTESTATION_AVAILABLE)), + "components_with_attestation_gaps": len(packages_with_attestation_gaps), + "components_with_enrichment_errors": sum(1 for component in provenance_components if _component_has_status(component, ProvenanceStatus.ENRICHMENT_ERROR)), + "unsupported_components": sum(1 for component in provenance_components if _component_has_status(component, ProvenanceStatus.UNSUPPORTED_FOR_PACKAGE)), + } + attestation_summary = { + "files_evaluated": sum(len(component.provenance.files) for component in provenance_components), + "files_with_attestations": sum( + 1 + for component in provenance_components + for file_evidence in component.provenance.files + if file_evidence.attestation_count > 0 + ), + "files_without_attestations": sum( + 1 + for component in provenance_components + for file_evidence in component.provenance.files + if file_evidence.attestation_count == 0 + ), + "packages_with_attestation_gaps": packages_with_attestation_gaps, + "publisher_kind_counts": {publisher: publisher_counts[publisher] for publisher in sorted(publisher_counts)}, + } + scorecard_results = [ + { + "component_key": component_key(component), + "name": component.name, + "version": component.version, + "repository": component.scorecard.repository.canonical_name if component.scorecard and component.scorecard.repository else None, + "repository_source": component.scorecard.repository.source if component.scorecard and component.scorecard.repository else None, + "status": _primary_scorecard_status(component), + "score": component.scorecard.score if component.scorecard else None, + "note": component.scorecard.note if component.scorecard else None, + "error": component.scorecard.error if component.scorecard else None, + } + for component in scorecard_components + ] + scorecard_summary = { + "enabled": report.metadata.enrichment.scorecard_enabled or bool(scorecard_components), + "components_in_scope": len(components), + "candidate_components": report.metadata.enrichment.scorecard_candidate_components, + "supported_components": report.metadata.enrichment.scorecard_supported_components, + "components_with_mapped_repositories": sum( + 1 for component in scorecard_components if component.scorecard and component.scorecard.repository is not None + ), + "components_with_scorecards": sum( + 1 for component in scorecard_components if _component_has_scorecard_status(component, ScorecardStatus.SCORECARD_AVAILABLE) + ), + "scorecard_unavailable": sum( + 1 for component in scorecard_components if _component_has_scorecard_status(component, ScorecardStatus.SCORECARD_UNAVAILABLE) + ), + "repository_unmapped": sum( + 1 for component in scorecard_components if _component_has_scorecard_status(component, ScorecardStatus.REPOSITORY_UNMAPPED) + ), + "components_with_enrichment_errors": sum( + 1 for component in scorecard_components if _component_has_scorecard_status(component, ScorecardStatus.ENRICHMENT_ERROR) + ), + "results": scorecard_results, + } + trust_signal_notes = _build_trust_signal_notes( + report, + provenance_components=provenance_components, + packages_with_attestation_gaps=packages_with_attestation_gaps, + publisher_counts=publisher_counts, + scorecard_components=scorecard_components, + ) + return { + "provenance_summary": provenance_summary, + "attestation_summary": attestation_summary, + "scorecard_summary": scorecard_summary, + "enrichment_metadata": enrichment_metadata_to_dict(report.metadata.enrichment), + "trust_signal_notes": trust_signal_notes, + } + + +def provenance_policy_violations(policy_evaluation: PolicyEvaluation | None) -> dict[str, list[PolicyViolation]]: + resolved = effective_policy_evaluation(policy_evaluation) + provenance_rule_ids = set(V2_PROVENANCE_POLICY_RULE_IDS) + return { + "blocking": [violation for violation in resolved.blocking_violations if violation.rule_id in provenance_rule_ids], + "warning": [violation for violation in resolved.warning_violations if violation.rule_id in provenance_rule_ids], + "suppressed": [violation for violation in resolved.suppressed_violations if violation.rule_id in provenance_rule_ids], + } + + +def provenance_policy_summary(policy_evaluation: PolicyEvaluation | None) -> dict[str, Any] | None: + resolved = effective_policy_evaluation(policy_evaluation) + policy = resolved.effective_policy + impacts = provenance_policy_violations(policy_evaluation) + if policy is None or not _policy_has_provenance_configuration(policy): + if not impacts["blocking"] and not impacts["warning"] and not impacts["suppressed"]: + return None + return { + "configured": False, + "requirements": { + "require_attestations_for_new_packages": False, + "require_provenance_for_suspicious_sources": False, + "allow_unattested_packages": [], + "allow_provenance_publishers": [], + }, + "counts": { + "blocking": len(impacts["blocking"]), + "warning": len(impacts["warning"]), + "suppressed": len(impacts["suppressed"]), + }, + "blocking": [policy_violation_to_dict(item) for item in impacts["blocking"]], + "warning": [policy_violation_to_dict(item) for item in impacts["warning"]], + "suppressed": [policy_violation_to_dict(item) for item in impacts["suppressed"]], + } + return { + "configured": True, + "requirements": { + "require_attestations_for_new_packages": policy.require_attestations_for_new_packages, + "require_provenance_for_suspicious_sources": policy.require_provenance_for_suspicious_sources, + "allow_unattested_packages": list(policy.allow_unattested_packages), + "allow_provenance_publishers": list(policy.allow_provenance_publishers), + }, + "counts": { + "blocking": len(impacts["blocking"]), + "warning": len(impacts["warning"]), + "suppressed": len(impacts["suppressed"]), + }, + "blocking": [policy_violation_to_dict(item) for item in impacts["blocking"]], + "warning": [policy_violation_to_dict(item) for item in impacts["warning"]], + "suppressed": [policy_violation_to_dict(item) for item in impacts["suppressed"]], + } + + +def scorecard_policy_violations(policy_evaluation: PolicyEvaluation | None) -> dict[str, list[PolicyViolation]]: + resolved = effective_policy_evaluation(policy_evaluation) + scorecard_rule_ids = set(V3_SCORECARD_POLICY_RULE_IDS) + return { + "blocking": [violation for violation in resolved.blocking_violations if violation.rule_id in scorecard_rule_ids], + "warning": [violation for violation in resolved.warning_violations if violation.rule_id in scorecard_rule_ids], + "suppressed": [violation for violation in resolved.suppressed_violations if violation.rule_id in scorecard_rule_ids], + } + + +def _policy_has_provenance_configuration(policy: PolicyConfig) -> bool: + return any( + ( + rule in V2_PROVENANCE_POLICY_RULE_IDS + for rule in (*policy.block_on, *policy.warn_on, *policy.ignore_rules) + ) + ) or policy.require_attestations_for_new_packages or policy.require_provenance_for_suspicious_sources or bool( + policy.allow_unattested_packages + ) or bool(policy.allow_provenance_publishers) + + +def _current_state_components(report: CompareReport) -> list[Component]: + components = list(report.components.added) + components.extend(change.after for change in report.components.changed) + return components + + +def _component_has_status(component: Component, status: ProvenanceStatus) -> bool: + if component.provenance is None: + return False + if status in component.provenance.statuses: + return True + return any(status in file_evidence.statuses for file_evidence in component.provenance.files) + + +def _component_has_attestation_gap(component: Component) -> bool: + if component.provenance is None: + return False + return _component_has_status(component, ProvenanceStatus.ATTESTATION_UNAVAILABLE) and not _component_has_status( + component, + ProvenanceStatus.ATTESTATION_AVAILABLE, + ) + + +def _component_has_scorecard_status(component: Component, status: ScorecardStatus) -> bool: + if component.scorecard is None: + return False + return status in component.scorecard.statuses + + +def _primary_scorecard_status(component: Component) -> str | None: + if component.scorecard is None or not component.scorecard.statuses: + return None + return component.scorecard.statuses[0].value + + +def _build_trust_signal_notes( + report: CompareReport, + *, + provenance_components: list[Component], + packages_with_attestation_gaps: list[dict[str, Any]], + publisher_counts: Counter[str], + scorecard_components: list[Component], +) -> list[str]: + notes: list[str] = [] + pypi_component_count = sum(1 for component in _current_state_components(report) if component.ecosystem.strip().lower() == "pypi") + unsupported_component_count = sum( + 1 for component in provenance_components if _component_has_status(component, ProvenanceStatus.UNSUPPORTED_FOR_PACKAGE) + ) + if not report.metadata.enrichment.pypi_enabled and not provenance_components: + if pypi_component_count: + notes.append("PyPI components are present, but provenance enrichment was not enabled for this run.") + else: + notes.append("No opt-in provenance enrichment data is present in this report.") + if packages_with_attestation_gaps: + notes.append( + "Missing attestations indicate an attestation gap for the release; they are not treated as proof of compromise." + ) + if any(_component_has_status(component, ProvenanceStatus.ENRICHMENT_ERROR) for component in provenance_components): + notes.append("PyPI enrichment errors are recorded as evidence gaps and only affect policy when configured explicitly.") + if unsupported_component_count: + notes.append( + "Some package versions could not provide provenance evidence from the enrichment source and remain evidence gaps." + ) + if publisher_counts: + notes.append(f"Observed attestation publisher kinds: {', '.join(sorted(publisher_counts))}.") + provenance_impacts = provenance_policy_violations(report.metadata.policy_evaluation) + impact_count = len(provenance_impacts["blocking"]) + len(provenance_impacts["warning"]) + if impact_count: + notes.append(f"Policy produced {impact_count} provenance-related blocking or warning decision(s).") + if report.metadata.enrichment.scorecard_enabled or scorecard_components: + notes.append("OpenSSF Scorecard results are auxiliary trust signals and are not proof of safety.") + if any(_component_has_scorecard_status(component, ScorecardStatus.REPOSITORY_UNMAPPED) for component in scorecard_components): + notes.append("Scorecard lookups are skipped when no high-confidence repository mapping is available.") + scorecard_impacts = scorecard_policy_violations(report.metadata.policy_evaluation) + scorecard_impact_count = len(scorecard_impacts["blocking"]) + len(scorecard_impacts["warning"]) + if scorecard_impact_count: + notes.append(f"Policy produced {scorecard_impact_count} Scorecard-related blocking or warning decision(s).") + return notes diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/pypi_integrity_client.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/pypi_integrity_client.py new file mode 100644 index 0000000..d6fb636 --- /dev/null +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/pypi_integrity_client.py @@ -0,0 +1,238 @@ +from __future__ import annotations + +import json +import socket +from dataclasses import dataclass +from typing import Any +from urllib import error, parse, request + + +@dataclass(slots=True, frozen=True) +class PyPIReleaseFile: + filename: str + url: str | None + sha256: str | None + upload_time: str | None + yanked: bool + + +@dataclass(slots=True, frozen=True) +class PyPIRelease: + project: str + version: str + release_url: str | None + files: tuple[PyPIReleaseFile, ...] + + +@dataclass(slots=True, frozen=True) +class PyPIAttestation: + statement: str | None + publisher_kind: str | None + + +@dataclass(slots=True, frozen=True) +class PyPIFileProvenance: + filename: str + attestation_count: int + attestations: tuple[PyPIAttestation, ...] + + +class PyPIClientError(RuntimeError): + def __init__( + self, + message: str, + *, + status_code: int | None = None, + is_timeout: bool = False, + ) -> None: + super().__init__(message) + self.status_code = status_code + self.is_timeout = is_timeout + + +class PyPIIntegrityClient: + def __init__( + self, + *, + timeout_seconds: float = 5.0, + base_url: str = "https://pypi.org", + opener: request.OpenerDirector | None = None, + ) -> None: + self.timeout_seconds = timeout_seconds + self.base_url = base_url.rstrip("/") + self._opener = opener or request.build_opener() + + def fetch_release(self, project: str, version: str) -> PyPIRelease: + encoded_project = parse.quote(project, safe="") + encoded_version = parse.quote(version, safe="") + payload = self._read_json(f"/pypi/{encoded_project}/{encoded_version}/json") + return parse_release_payload(payload, project=project, version=version) + + def fetch_provenance(self, project: str, version: str, filename: str) -> PyPIFileProvenance | None: + encoded_project = parse.quote(project, safe="") + encoded_version = parse.quote(version, safe="") + encoded_filename = parse.quote(filename, safe="") + path = f"/integrity/{encoded_project}/{encoded_version}/{encoded_filename}/provenance" + try: + payload = self._read_json(path) + except PyPIClientError as exc: + if exc.status_code == 404: + return None + raise + return parse_provenance_payload(payload, filename=filename) + + def _read_json(self, path: str) -> object: + url = f"{self.base_url}{path}" + req = request.Request( + url, + headers={ + "Accept": "application/json", + "User-Agent": "sbom-diff-and-risk pypi-integrity-client", + }, + ) + try: + with self._opener.open(req, timeout=self.timeout_seconds) as response: + payload = json.load(response) + except error.HTTPError as exc: + raise PyPIClientError( + f"PyPI request failed with HTTP {exc.code} for {url}.", + status_code=exc.code, + ) from exc + except error.URLError as exc: + if _is_timeout_reason(exc.reason): + raise PyPIClientError( + f"PyPI request timed out after {self.timeout_seconds} seconds for {url}.", + is_timeout=True, + ) from exc + raise PyPIClientError(f"PyPI request failed for {url}: {exc.reason}.") from exc + except TimeoutError as exc: + raise PyPIClientError( + f"PyPI request timed out after {self.timeout_seconds} seconds for {url}.", + is_timeout=True, + ) from exc + except socket.timeout as exc: + raise PyPIClientError( + f"PyPI request timed out after {self.timeout_seconds} seconds for {url}.", + is_timeout=True, + ) from exc + except json.JSONDecodeError as exc: + raise PyPIClientError( + f"PyPI returned malformed JSON for {url}: line {exc.lineno}, column {exc.colno}: {exc.msg}." + ) from exc + + return payload + + +def parse_release_payload(payload: object, *, project: str, version: str) -> PyPIRelease: + if not isinstance(payload, dict): + raise PyPIClientError("PyPI release response must be a JSON object.") + + raw_info = payload.get("info") + if raw_info is None or not isinstance(raw_info, dict): + raise PyPIClientError("PyPI release response is missing an info object.") + + release_url = _optional_text(raw_info.get("release_url")) or _optional_text(raw_info.get("package_url")) + raw_files = payload.get("urls") + if raw_files is None: + raw_files = [] + if not isinstance(raw_files, list): + raise PyPIClientError("PyPI release response urls field must be a list.") + + files: list[PyPIReleaseFile] = [] + for raw_file in raw_files: + if not isinstance(raw_file, dict): + raise PyPIClientError("PyPI release file entries must be JSON objects.") + filename = _required_text(raw_file.get("filename"), "PyPI release file filename") + raw_digests = raw_file.get("digests") + if raw_digests is not None and not isinstance(raw_digests, dict): + raise PyPIClientError("PyPI release file digests field must be an object when present.") + sha256 = None + if isinstance(raw_digests, dict): + sha256 = _optional_text(raw_digests.get("sha256")) + files.append( + PyPIReleaseFile( + filename=filename, + url=_optional_text(raw_file.get("url")), + sha256=sha256, + upload_time=_optional_text(raw_file.get("upload_time_iso_8601")), + yanked=bool(raw_file.get("yanked", False)), + ) + ) + + files.sort(key=lambda item: item.filename.lower()) + return PyPIRelease( + project=project, + version=version, + release_url=release_url, + files=tuple(files), + ) + + +def parse_provenance_payload(payload: object, *, filename: str) -> PyPIFileProvenance: + if not isinstance(payload, dict): + raise PyPIClientError("PyPI provenance response must be a JSON object.") + + raw_bundles = payload.get("attestation_bundles") + if raw_bundles is None: + raw_bundles = [] + if not isinstance(raw_bundles, list): + raise PyPIClientError("PyPI provenance response attestation_bundles field must be a list.") + + attestations: list[PyPIAttestation] = [] + for raw_bundle in raw_bundles: + if not isinstance(raw_bundle, dict): + raise PyPIClientError("PyPI provenance bundles must be JSON objects.") + publisher_kind = None + raw_publisher = raw_bundle.get("publisher") + if raw_publisher is not None: + if not isinstance(raw_publisher, dict): + raise PyPIClientError("PyPI provenance publisher must be an object when present.") + publisher_kind = _optional_text(raw_publisher.get("kind")) + + raw_attestations = raw_bundle.get("attestations") + if raw_attestations is None: + raw_attestations = [] + if not isinstance(raw_attestations, list): + raise PyPIClientError("PyPI provenance bundle attestations field must be a list.") + + for raw_attestation in raw_attestations: + if not isinstance(raw_attestation, dict): + raise PyPIClientError("PyPI provenance attestations must be JSON objects.") + raw_envelope = raw_attestation.get("envelope") + if raw_envelope is None or not isinstance(raw_envelope, dict): + raise PyPIClientError("PyPI provenance attestation envelope must be an object.") + attestations.append( + PyPIAttestation( + statement=_optional_text(raw_envelope.get("statement")), + publisher_kind=publisher_kind, + ) + ) + + return PyPIFileProvenance( + filename=filename, + attestation_count=len(attestations), + attestations=tuple(attestations), + ) + + +def _required_text(value: object, context: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise PyPIClientError(f"{context} must be a non-empty string.") + return value + + +def _optional_text(value: object) -> str | None: + if value is None: + return None + if not isinstance(value, str): + raise PyPIClientError("Expected a string value in PyPI response.") + stripped = value.strip() + return stripped or None + + +def _is_timeout_reason(reason: object) -> bool: + if isinstance(reason, (TimeoutError, socket.timeout)): + return True + if isinstance(reason, str): + return "timed out" in reason.lower() + return False diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/pypi_provenance.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/pypi_provenance.py new file mode 100644 index 0000000..e8b2f56 --- /dev/null +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/pypi_provenance.py @@ -0,0 +1,237 @@ +from __future__ import annotations + +import base64 +import binascii +import json + +from .models import ( + Component, + ProvenanceEvidence, + ProvenanceFileEvidence, + ProvenanceStatus, +) +from .pypi_integrity_client import ( + PyPIClientError, + PyPIFileProvenance, + PyPIIntegrityClient, + PyPIRelease, +) + +_STATUS_ORDER = { + ProvenanceStatus.PROVENANCE_AVAILABLE: 0, + ProvenanceStatus.ATTESTATION_AVAILABLE: 1, + ProvenanceStatus.ATTESTATION_UNAVAILABLE: 2, + ProvenanceStatus.ENRICHMENT_ERROR: 3, + ProvenanceStatus.UNSUPPORTED_FOR_PACKAGE: 4, +} + + +def normalize_pypi_provenance(component: Component, *, client: PyPIIntegrityClient) -> ProvenanceEvidence: + if component.ecosystem.strip().lower() != "pypi": + return _unsupported_provenance(component) + if not component.name.strip() or not component.version or not component.version.strip(): + return _unsupported_provenance(component) + + try: + release = client.fetch_release(component.name, component.version) + except PyPIClientError as exc: + if exc.status_code == 404: + return _unsupported_provenance(component, lookup_performed=True) + return ProvenanceEvidence( + provider="pypi", + requested=True, + supported=True, + lookup_performed=True, + package_name=component.name, + package_version=component.version, + release_url=_release_url(component.name, component.version), + statuses=(ProvenanceStatus.ENRICHMENT_ERROR,), + error=str(exc), + ) + + return _normalize_release_provenance(component, release=release, client=client) + + +def normalize_provenance_file( + *, + release_file, + provenance: PyPIFileProvenance | None, +) -> ProvenanceFileEvidence: + if provenance is None: + return ProvenanceFileEvidence( + filename=release_file.filename, + url=release_file.url, + sha256=release_file.sha256, + upload_time=release_file.upload_time, + yanked=release_file.yanked, + statuses=(ProvenanceStatus.ATTESTATION_UNAVAILABLE,), + attestation_count=0, + ) + + predicate_types = sorted( + { + predicate_type + for predicate_type in (_decode_statement_predicate_type(item.statement) for item in provenance.attestations) + if predicate_type + } + ) + publisher_kinds = sorted({item.publisher_kind for item in provenance.attestations if item.publisher_kind}) + statuses = [ProvenanceStatus.PROVENANCE_AVAILABLE] + if provenance.attestation_count > 0: + statuses.append(ProvenanceStatus.ATTESTATION_AVAILABLE) + else: + statuses.append(ProvenanceStatus.ATTESTATION_UNAVAILABLE) + + return ProvenanceFileEvidence( + filename=release_file.filename, + url=release_file.url, + sha256=release_file.sha256, + upload_time=release_file.upload_time, + yanked=release_file.yanked, + statuses=tuple(statuses), + attestation_count=provenance.attestation_count, + predicate_types=tuple(predicate_types), + publisher_kinds=tuple(publisher_kinds), + ) + + +def provenance_evidence_to_dict(provenance: ProvenanceEvidence | None) -> dict[str, object] | None: + if provenance is None: + return None + return { + "provider": provenance.provider, + "requested": provenance.requested, + "supported": provenance.supported, + "lookup_performed": provenance.lookup_performed, + "package_name": provenance.package_name, + "package_version": provenance.package_version, + "release_url": provenance.release_url, + "statuses": [status.value for status in provenance.statuses], + "files": [ + { + "filename": item.filename, + "url": item.url, + "sha256": item.sha256, + "upload_time": item.upload_time, + "yanked": item.yanked, + "statuses": [status.value for status in item.statuses], + "attestation_count": item.attestation_count, + "predicate_types": list(item.predicate_types), + "publisher_kinds": list(item.publisher_kinds), + "error": item.error, + } + for item in provenance.files + ], + "files_evaluated": provenance.files_evaluated, + "files_with_attestations": provenance.files_with_attestations, + "files_without_attestations": provenance.files_without_attestations, + "error": provenance.error, + } + + +def _normalize_release_provenance( + component: Component, + *, + release: PyPIRelease, + client: PyPIIntegrityClient, +) -> ProvenanceEvidence: + if not release.files: + return _unsupported_provenance(component, lookup_performed=True) + + component_statuses: set[ProvenanceStatus] = set() + file_evidence: list[ProvenanceFileEvidence] = [] + first_error: str | None = None + + for release_file in release.files: + try: + provenance = client.fetch_provenance(component.name, component.version or "", release_file.filename) + except PyPIClientError as exc: + component_statuses.add(ProvenanceStatus.ENRICHMENT_ERROR) + if first_error is None: + first_error = str(exc) + file_evidence.append( + ProvenanceFileEvidence( + filename=release_file.filename, + url=release_file.url, + sha256=release_file.sha256, + upload_time=release_file.upload_time, + yanked=release_file.yanked, + statuses=(ProvenanceStatus.ENRICHMENT_ERROR,), + error=str(exc), + ) + ) + continue + + normalized_file = normalize_provenance_file(release_file=release_file, provenance=provenance) + file_evidence.append(normalized_file) + component_statuses.update(normalized_file.statuses) + + files_evaluated = len(file_evidence) + files_with_attestations = sum(1 for item in file_evidence if item.attestation_count > 0) + return ProvenanceEvidence( + provider="pypi", + requested=True, + supported=True, + lookup_performed=True, + package_name=component.name, + package_version=component.version, + release_url=release.release_url or _release_url(component.name, component.version or ""), + statuses=_sorted_statuses(component_statuses or {ProvenanceStatus.ATTESTATION_UNAVAILABLE}), + files=tuple(file_evidence), + files_evaluated=files_evaluated, + files_with_attestations=files_with_attestations, + files_without_attestations=files_evaluated - files_with_attestations, + error=first_error, + ) + + +def _unsupported_provenance( + component: Component, + *, + lookup_performed: bool = False, +) -> ProvenanceEvidence: + return ProvenanceEvidence( + provider="pypi", + requested=True, + supported=False, + lookup_performed=lookup_performed, + package_name=component.name, + package_version=component.version, + release_url=_release_url(component.name, component.version or ""), + statuses=(ProvenanceStatus.UNSUPPORTED_FOR_PACKAGE,), + ) + + +def _release_url(name: str, version: str) -> str: + if version: + return f"https://pypi.org/project/{name}/{version}/" + return f"https://pypi.org/project/{name}/" + + +def _decode_statement_predicate_type(statement: str | None) -> str | None: + if not statement: + return None + + padding = (-len(statement)) % 4 + encoded = statement + ("=" * padding) + try: + decoded = base64.urlsafe_b64decode(encoded.encode("utf-8")) + except (ValueError, binascii.Error): + return None + + try: + payload = json.loads(decoded) + except json.JSONDecodeError: + return None + if not isinstance(payload, dict): + return None + + predicate_type = payload.get("predicateType") + if not isinstance(predicate_type, str): + return None + stripped = predicate_type.strip() + return stripped or None + + +def _sorted_statuses(statuses: set[ProvenanceStatus]) -> tuple[ProvenanceStatus, ...]: + return tuple(sorted(statuses, key=lambda item: (_STATUS_ORDER[item], item.value))) diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_json.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_json.py index 2026dae..2324103 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_json.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_json.py @@ -2,12 +2,15 @@ import json +from .enrichment import enrichment_metadata_to_dict, provenance_evidence_to_dict from .models import CompareReport, Component, ComponentChange, RiskFinding -from .presentation import build_policy_report_sections +from .presentation import build_policy_report_sections, build_trust_signal_report_sections +from .scorecard_enrichment import scorecard_evidence_to_dict def render_report_json(report: CompareReport) -> str: policy_sections = build_policy_report_sections(report.metadata.policy_evaluation) + trust_signal_sections = build_trust_signal_report_sections(report) payload = { "summary": { "added": report.summary.added, @@ -26,6 +29,11 @@ def render_report_json(report: CompareReport) -> str: "warning_findings": policy_sections["warning_findings"], "suppressed_findings": policy_sections["suppressed_findings"], "rule_catalog": policy_sections["rule_catalog"], + "provenance_summary": trust_signal_sections["provenance_summary"], + "attestation_summary": trust_signal_sections["attestation_summary"], + "scorecard_summary": trust_signal_sections["scorecard_summary"], + "enrichment_metadata": trust_signal_sections["enrichment_metadata"], + "trust_signal_notes": trust_signal_sections["trust_signal_notes"], "metadata": { "before_format": report.metadata.before_format, "after_format": report.metadata.after_format, @@ -33,13 +41,24 @@ def render_report_json(report: CompareReport) -> str: "strict": report.metadata.strict, "stub": report.metadata.stub, "policy_evaluation": policy_sections["policy_evaluation"], + "enrichment": enrichment_metadata_to_dict(report.metadata.enrichment), }, "notes": list(report.notes), } + if policy_sections["provenance_policy"] is not None: + payload["provenance_policy"] = policy_sections["provenance_policy"] + payload["provenance_policy_impact"] = policy_sections["provenance_policy_impact"] return json.dumps(payload, indent=2) + "\n" def _component_to_dict(component: Component) -> dict[str, object]: + evidence = dict(component.evidence) + provenance = provenance_evidence_to_dict(component.provenance) + if provenance is not None: + evidence["provenance"] = provenance + scorecard = scorecard_evidence_to_dict(component.scorecard) + if scorecard is not None: + evidence["scorecard"] = scorecard return { "name": component.name, "version": component.version, @@ -50,7 +69,7 @@ def _component_to_dict(component: Component) -> dict[str, object]: "source_url": component.source_url, "bom_ref": component.bom_ref, "raw_type": component.raw_type, - "evidence": component.evidence, + "evidence": evidence, } diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_md.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_md.py index 06416df..920947c 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_md.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_md.py @@ -2,11 +2,25 @@ from .diffing import component_key from .models import CompareReport -from .presentation import effective_policy_evaluation +from .presentation import ( + build_trust_signal_report_sections, + effective_policy_evaluation, + provenance_policy_summary, + provenance_policy_violations, + scorecard_policy_violations, +) def render_report_markdown(report: CompareReport) -> str: policy_evaluation = effective_policy_evaluation(report.metadata.policy_evaluation) + trust_signal_sections = build_trust_signal_report_sections(report) + provenance_impacts = provenance_policy_violations(report.metadata.policy_evaluation) + provenance_policy = provenance_policy_summary(report.metadata.policy_evaluation) + scorecard_impacts = scorecard_policy_violations(report.metadata.policy_evaluation) + provenance_summary = trust_signal_sections["provenance_summary"] + attestation_summary = trust_signal_sections["attestation_summary"] + scorecard_summary = trust_signal_sections["scorecard_summary"] + enrichment_metadata = trust_signal_sections["enrichment_metadata"] lines = [ "# sbom-diff-and-risk report", "", @@ -36,6 +50,145 @@ def render_report_markdown(report: CompareReport) -> str: ] ) + lines.extend( + [ + "", + "## Provenance summary", + f"- Enrichment mode: {enrichment_metadata['mode']}", + f"- Network access performed: {'yes' if enrichment_metadata['pypi_network_access_performed'] else 'no'}", + f"- Candidate components for enrichment: {enrichment_metadata['candidate_components']}", + f"- Supported components for enrichment: {enrichment_metadata['supported_components']}", + f"- Observed provenance status counts: {_format_status_counts(enrichment_metadata['status_counts'])}", + f"- Components in scope: {provenance_summary['components_in_scope']}", + f"- PyPI components in scope: {provenance_summary['pypi_components_in_scope']}", + f"- PyPI components without provenance records: {provenance_summary['pypi_components_without_provenance']}", + f"- Components with provenance evidence: {provenance_summary['components_with_provenance']}", + f"- Components with attestations: {provenance_summary['components_with_attestations']}", + f"- Components with attestation gaps: {provenance_summary['components_with_attestation_gaps']}", + f"- Components with enrichment errors: {provenance_summary['components_with_enrichment_errors']}", + f"- Unsupported components: {provenance_summary['unsupported_components']}", + "", + "## Attestation gaps", + "| component | version | statuses |", + "|-----------|---------|----------|", + ] + ) + if attestation_summary["packages_with_attestation_gaps"]: + for package in attestation_summary["packages_with_attestation_gaps"]: + lines.append( + f"| {package['name']} | {package['version'] or ''} | " + f"{_escape_table_text(', '.join(package['statuses']))} |" + ) + else: + lines.append("| _none_ | | |") + + lines.extend( + [ + "", + "## Policy impact for provenance-related rules", + ] + ) + if provenance_policy is not None: + lines.extend( + [ + f"- Configured provenance policy: {'yes' if provenance_policy['configured'] else 'no'}", + ( + "- Require attestations for new packages: yes" + if provenance_policy["requirements"]["require_attestations_for_new_packages"] + else "- Require attestations for new packages: no" + ), + ( + "- Require provenance for suspicious sources: yes" + if provenance_policy["requirements"]["require_provenance_for_suspicious_sources"] + else "- Require provenance for suspicious sources: no" + ), + ( + f"- Allow unattested packages: {', '.join(provenance_policy['requirements']['allow_unattested_packages'])}" + if provenance_policy["requirements"]["allow_unattested_packages"] + else "- Allow unattested packages: none" + ), + ( + f"- Allowed provenance publishers: {', '.join(provenance_policy['requirements']['allow_provenance_publishers'])}" + if provenance_policy["requirements"]["allow_provenance_publishers"] + else "- Allowed provenance publishers: none" + ), + ( + f"- Provenance policy decisions: blocking={provenance_policy['counts']['blocking']}, " + f"warning={provenance_policy['counts']['warning']}, " + f"suppressed={provenance_policy['counts']['suppressed']}" + ), + ] + ) + lines.extend( + [ + "| rule id | component | level | message |", + "|---------|-----------|-------|---------|", + ] + ) + provenance_violations = [*provenance_impacts["blocking"], *provenance_impacts["warning"]] + if provenance_violations: + for violation in provenance_violations: + lines.append( + f"| {violation.rule_id} | {violation.component_name or ''} | " + f"{violation.level.value if violation.level else ''} | {_escape_table_text(violation.message)} |" + ) + else: + lines.append("| _none_ | | | |") + + lines.extend(["", "## Trust signal notes"]) + if trust_signal_sections["trust_signal_notes"]: + lines.extend(f"- {note}" for note in trust_signal_sections["trust_signal_notes"]) + else: + lines.append("- No additional trust signal notes.") + + lines.extend( + [ + "", + "## Scorecard summary", + f"- Enrichment enabled: {'yes' if scorecard_summary['enabled'] else 'no'}", + f"- Network access performed: {'yes' if enrichment_metadata['scorecard_network_access_performed'] else 'no'}", + f"- Candidate components for Scorecard enrichment: {scorecard_summary['candidate_components']}", + f"- Components with supported repository mappings: {scorecard_summary['supported_components']}", + f"- Components with mapped repositories: {scorecard_summary['components_with_mapped_repositories']}", + f"- Components with available Scorecards: {scorecard_summary['components_with_scorecards']}", + f"- Scorecard unavailable: {scorecard_summary['scorecard_unavailable']}", + f"- Repository unmapped: {scorecard_summary['repository_unmapped']}", + f"- Components with enrichment errors: {scorecard_summary['components_with_enrichment_errors']}", + f"- Observed Scorecard status counts: {_format_status_counts(enrichment_metadata['scorecard_status_counts'])}", + "", + "## Scorecard results", + "| component | version | repository | score | status |", + "|-----------|---------|------------|-------|--------|", + ] + ) + if scorecard_summary["results"]: + for result in scorecard_summary["results"]: + score = "" if result["score"] is None else f"{result['score']:.1f}" + lines.append( + f"| {result['name']} | {result['version'] or ''} | {result['repository'] or ''} | " + f"{score} | {_escape_table_text(result['status'] or '')} |" + ) + else: + lines.append("| _none_ | | | | |") + + lines.extend( + [ + "", + "## Policy impact for Scorecard-related rules", + "| rule id | component | level | message |", + "|---------|-----------|-------|---------|", + ] + ) + scorecard_violations = [*scorecard_impacts["blocking"], *scorecard_impacts["warning"]] + if scorecard_violations: + for violation in scorecard_violations: + lines.append( + f"| {violation.rule_id} | {violation.component_name or ''} | " + f"{violation.level.value if violation.level else ''} | {_escape_table_text(violation.message)} |" + ) + else: + lines.append("| _none_ | | | |") + lines.extend( [ "", @@ -175,3 +328,9 @@ def _risk_labels_for_component(report: CompareReport, component) -> str: def _escape_table_text(value: str) -> str: return value.replace("|", "\\|") + + +def _format_status_counts(status_counts: dict[str, int]) -> str: + if not status_counts: + return "none" + return ", ".join(f"{key}={value}" for key, value in status_counts.items()) diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_sarif.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_sarif.py index 14bcb19..0a4f90d 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_sarif.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_sarif.py @@ -7,7 +7,7 @@ from . import __version__ from .models import CompareReport, RiskBucket, RiskFinding -from .policy_models import PolicyViolation +from .policy_models import PolicyLevel, PolicyViolation from .presentation import effective_policy_evaluation, rule_catalog_to_dict DEFAULT_SARIF_RESULT_LIMIT = 5000 @@ -21,14 +21,35 @@ RiskBucket.UNKNOWN_LICENSE, RiskBucket.MAJOR_UPGRADE, } -_SARIF_POLICY_ONLY_RULE_IDS = {"allow_sources", "max_added_packages"} +_SARIF_POLICY_ONLY_RULE_IDS = { + "allow_sources", + "max_added_packages", + "provenance_required", + "missing_attestation", + "unverified_provenance", + "scorecard_below_threshold", +} +_SARIF_HIGH_SIGNAL_PROVENANCE_RULE_IDS = { + "provenance_required", + "missing_attestation", + "unverified_provenance", +} _LEVEL_PRIORITY = {"error": 0, "warning": 1, "note": 2} +_POLICY_LEVEL_PRIORITY = { + PolicyLevel.BLOCK: 0, + PolicyLevel.WARN: 1, + None: 99, +} _RULE_PRIORITY = { "sdr.suspicious_source": 0, "sdr.unknown_license": 1, "sdr.major_upgrade": 2, - "sdr.policy_violation.allow_sources": 3, - "sdr.policy_violation.max_added_packages": 4, + "sdr.policy_violation.provenance_required": 3, + "sdr.policy_violation.unverified_provenance": 4, + "sdr.policy_violation.missing_attestation": 5, + "sdr.policy_violation.allow_sources": 6, + "sdr.policy_violation.max_added_packages": 7, + "sdr.policy_violation.scorecard_below_threshold": 8, } @@ -91,6 +112,7 @@ def render_report_sarif_output( resolved_base_dir = base_dir.resolve() if base_dir is not None else None policy_evaluation = effective_policy_evaluation(report.metadata.policy_evaluation) blocking_map = _blocking_violation_map(policy_evaluation.blocking_violations) + provenance_required_levels = _provenance_required_levels(policy_evaluation) emitted_blocking_keys: set[tuple[str, str | None]] = set() candidate_results: list[dict[str, Any]] = [] @@ -98,6 +120,8 @@ def render_report_sarif_output( for finding in report.risks: if finding.bucket not in _SARIF_SUPPORTED_RISK_BUCKETS: continue + if finding.bucket is RiskBucket.SUSPICIOUS_SOURCE and finding.component_key in provenance_required_levels: + continue policy_rule_id = _policy_rule_id_for_bucket(finding.bucket) blocking_violation = blocking_map.get((policy_rule_id, finding.component_key)) @@ -111,7 +135,7 @@ def render_report_sarif_output( if blocking_violation is not None: emitted_blocking_keys.add((policy_rule_id, finding.component_key)) - for violation in policy_evaluation.blocking_violations: + for violation in _eligible_policy_violations(policy_evaluation): lookup_key = (violation.rule_id, violation.component_key) if lookup_key in emitted_blocking_keys: continue @@ -239,7 +263,7 @@ def _policy_violation_to_result( return { "ruleId": rule_id, - "level": "error", + "level": _policy_result_level(violation), "message": { "text": _policy_result_message(violation), }, @@ -287,15 +311,71 @@ def _policy_result_message(violation: PolicyViolation) -> str: if violation.rule_id == "allow_sources" and violation.component_name: component_label = _component_label(violation.component_name, None) return f"{component_label}: {violation.message}" + if violation.rule_id == "missing_attestation" and violation.component_name: + return _component_policy_message(violation.component_name, "No PyPI attestations were published for this release.") + if violation.rule_id == "unverified_provenance" and violation.component_name: + return _component_policy_message( + violation.component_name, + "PyPI attestation publisher could not be verified by policy.", + ) + if violation.rule_id == "provenance_required" and violation.component_name: + return _component_policy_message( + violation.component_name, + _concise_provenance_required_message(violation.message), + ) + if violation.rule_id == "scorecard_below_threshold" and violation.component_name: + return _component_policy_message( + violation.component_name, + violation.message, + ) return violation.message +def _policy_result_level(violation: PolicyViolation) -> str: + if violation.level is PolicyLevel.WARN: + return "warning" + return "error" + + def _component_label(name: str, version: str | None) -> str: if version: return f"{name} {version}" return name +def _component_policy_message(component_name: str, message: str) -> str: + return f"{_component_label(component_name, None)}: {message}" + + +def _concise_provenance_required_message(message: str) -> str: + lowered = message.lower() + context = _provenance_requirement_context(message) + + if "no attestations were published" in lowered: + reason = "no attestations were published" + elif "could not be verified" in lowered: + reason = "available attestations could not be verified" + elif "evidence is unavailable" in lowered or "evidence was unavailable" in lowered: + reason = "provenance evidence was unavailable" + else: + normalized = message.rstrip(".") + return normalized[0].upper() + normalized[1:] + "." + + if context: + return f"Provenance required for {context}; {reason}." + return f"Provenance required; {reason}." + + +def _provenance_requirement_context(message: str) -> str | None: + prefix = "Provenance is required for " + delimiter = ", but" + if not message.startswith(prefix): + return None + context, _, _ = message[len(prefix):].partition(delimiter) + normalized = context.strip() + return normalized or None + + def _blocking_violation_map(violations: list[PolicyViolation]) -> dict[tuple[str, str | None], PolicyViolation]: return { (violation.rule_id, violation.component_key): violation @@ -303,6 +383,52 @@ def _blocking_violation_map(violations: list[PolicyViolation]) -> dict[tuple[str } +def _provenance_required_levels(policy_evaluation) -> dict[str | None, PolicyLevel | None]: + return { + violation.component_key: violation.level + for violation in (*policy_evaluation.blocking_violations, *policy_evaluation.warning_violations) + if violation.rule_id == "provenance_required" + } + + +def _eligible_policy_violations(policy_evaluation) -> list[PolicyViolation]: + provenance_required_levels = _provenance_required_levels(policy_evaluation) + eligible: list[PolicyViolation] = [] + for violation in policy_evaluation.blocking_violations: + if sarif_rule_id_for_policy_violation(violation.rule_id) is not None and _should_emit_policy_violation( + violation, + provenance_required_levels=provenance_required_levels, + ): + eligible.append(violation) + for violation in policy_evaluation.warning_violations: + if sarif_rule_id_for_policy_violation(violation.rule_id) is None: + continue + if _should_emit_policy_violation(violation, provenance_required_levels=provenance_required_levels): + eligible.append(violation) + return eligible + + +def _should_emit_policy_violation( + violation: PolicyViolation, + *, + provenance_required_levels: dict[str | None, PolicyLevel | None], +) -> bool: + if violation.rule_id in {"allow_sources", "max_added_packages", "provenance_required", "scorecard_below_threshold"}: + return True + if violation.rule_id in {"missing_attestation", "unverified_provenance"}: + provenance_required_level = provenance_required_levels.get(violation.component_key) + if provenance_required_level is not None and _policy_level_rank(provenance_required_level) <= _policy_level_rank( + violation.level + ): + return False + return violation.level is PolicyLevel.BLOCK + return False + + +def _policy_level_rank(level: PolicyLevel | None) -> int: + return _POLICY_LEVEL_PRIORITY[level] + + def _policy_rule_id_for_bucket(bucket: RiskBucket) -> str: return bucket.value @@ -380,20 +506,26 @@ def _sarif_rule_metadata(rule_id: str) -> dict[str, Any]: if rule_id.startswith("sdr.policy_violation."): policy_rule_id = rule_id.removeprefix("sdr.policy_violation.") description = catalog.get(policy_rule_id, {}).get("description", "Blocking policy violation.") + default_level = "warning" if policy_rule_id in {"missing_attestation", "scorecard_below_threshold"} else "error" + tags = ["supply-chain", "policy"] + if policy_rule_id in _SARIF_HIGH_SIGNAL_PROVENANCE_RULE_IDS: + tags.append("provenance") + if policy_rule_id == "scorecard_below_threshold": + tags.append("scorecard") return { "id": rule_id, "name": f"policy_violation.{policy_rule_id}", "shortDescription": { - "text": f"Blocking policy violation: {policy_rule_id}", + "text": f"Policy violation: {policy_rule_id}", }, "fullDescription": { "text": description, }, "defaultConfiguration": { - "level": "error", + "level": default_level, }, "properties": { - "tags": ["supply-chain", "policy"], + "tags": tags, }, } diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/repository_mapping.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/repository_mapping.py new file mode 100644 index 0000000..4956caa --- /dev/null +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/repository_mapping.py @@ -0,0 +1,248 @@ +from __future__ import annotations + +from dataclasses import dataclass +import re +from typing import Iterable +from urllib.parse import urlparse + +from .models import Component, RepositoryMapping, RepositoryMappingConfidence + +_SUPPORTED_SCORECARD_PLATFORMS = {"github.com"} +_REF_SOURCE_PRIORITY = { + "cyclonedx.externalReferences.vcs": 0, + "spdx.externalRefs.vcs": 1, + "component.source_url": 2, + "cyclonedx.externalReferences.website": 10, + "cyclonedx.externalReferences.distribution": 10, + "spdx.externalRefs": 11, + "spdx.homepage": 12, + "spdx.downloadLocation": 12, +} +_HIGH_CONFIDENCE_SOURCES = { + "cyclonedx.externalReferences.vcs", + "spdx.externalRefs.vcs", + "component.source_url", +} + + +@dataclass(slots=True, frozen=True) +class RepositoryMappingCandidate: + mapping: RepositoryMapping + priority: int + + +@dataclass(slots=True, frozen=True) +class RepositoryMappingAssessment: + mapping: RepositoryMapping | None + confidence: RepositoryMappingConfidence | None + reason: str + candidates: tuple[RepositoryMappingCandidate, ...] = () + + +def map_component_to_repository(component: Component) -> RepositoryMapping | None: + return assess_component_repository_mapping(component).mapping + + +def assess_component_repository_mapping(component: Component) -> RepositoryMappingAssessment: + candidates = tuple(_repository_candidates(component)) + if not candidates: + return RepositoryMappingAssessment( + mapping=None, + confidence=None, + reason="no_repository_candidates", + ) + + high_confidence_candidates = tuple( + candidate + for candidate in candidates + if candidate.mapping.confidence is RepositoryMappingConfidence.HIGH + ) + if not high_confidence_candidates: + return RepositoryMappingAssessment( + mapping=None, + confidence=RepositoryMappingConfidence.LOW, + reason="only_low_confidence_candidates", + candidates=candidates, + ) + + canonical_names = {candidate.mapping.canonical_name for candidate in high_confidence_candidates} + if len(canonical_names) != 1: + return RepositoryMappingAssessment( + mapping=None, + confidence=RepositoryMappingConfidence.HIGH, + reason="ambiguous_high_confidence_candidates", + candidates=high_confidence_candidates, + ) + + selected = min( + high_confidence_candidates, + key=lambda candidate: ( + candidate.priority, + candidate.mapping.source, + candidate.mapping.canonical_name, + ), + ) + return RepositoryMappingAssessment( + mapping=selected.mapping, + confidence=selected.mapping.confidence, + reason="mapped", + candidates=high_confidence_candidates, + ) + + +def repository_mapping_cache_key(component: Component) -> tuple[str, str, str, tuple[tuple[str, str], ...]]: + return ( + component.ecosystem.strip().lower(), + component.name.strip().lower(), + (component.version or "").strip().lower(), + tuple(sorted((source, raw_url.strip()) for raw_url, source in _candidate_urls(component))), + ) + + +def _repository_candidates(component: Component) -> list[RepositoryMappingCandidate]: + candidates: list[RepositoryMappingCandidate] = [] + for raw_url, source in _candidate_urls(component): + mapping = _normalize_repository_url(raw_url, source=source) + if mapping is None: + continue + candidates.append( + RepositoryMappingCandidate( + mapping=mapping, + priority=_REF_SOURCE_PRIORITY.get(source, 99), + ) + ) + return _dedupe_candidates(candidates) + + +def _candidate_urls(component: Component) -> Iterable[tuple[str, str]]: + source_format = component.evidence.get("source_format") + if source_format == "cyclonedx-json": + raw_component = component.evidence.get("component") + if isinstance(raw_component, dict): + yield from _cyclonedx_reference_urls(raw_component) + return + elif source_format == "spdx-json": + raw_package = component.evidence.get("package") + if isinstance(raw_package, dict): + yield from _spdx_reference_urls(raw_package) + return + + if component.source_url: + yield component.source_url, "component.source_url" + + +def _cyclonedx_reference_urls(raw_component: dict[str, object]) -> Iterable[tuple[str, str]]: + raw_refs = raw_component.get("externalReferences") + if not isinstance(raw_refs, list): + return () + + urls: list[tuple[str, str]] = [] + for raw_ref in raw_refs: + if not isinstance(raw_ref, dict): + continue + raw_url = raw_ref.get("url") + raw_type = raw_ref.get("type") + if not isinstance(raw_url, str) or not raw_url.strip(): + continue + if raw_type == "vcs": + urls.append((raw_url, "cyclonedx.externalReferences.vcs")) + elif raw_type == "website": + urls.append((raw_url, "cyclonedx.externalReferences.website")) + elif raw_type == "distribution": + urls.append((raw_url, "cyclonedx.externalReferences.distribution")) + return tuple(urls) + + +def _spdx_reference_urls(raw_package: dict[str, object]) -> Iterable[tuple[str, str]]: + urls: list[tuple[str, str]] = [] + + homepage = raw_package.get("homepage") + if isinstance(homepage, str) and homepage.strip() and homepage != "NOASSERTION": + urls.append((homepage, "spdx.homepage")) + + download_location = raw_package.get("downloadLocation") + if isinstance(download_location, str) and download_location.strip() and download_location != "NOASSERTION": + urls.append((download_location, "spdx.downloadLocation")) + + raw_refs = raw_package.get("externalRefs") + if not isinstance(raw_refs, list): + return tuple(urls) + + for raw_ref in raw_refs: + if not isinstance(raw_ref, dict): + continue + reference_type = raw_ref.get("referenceType") + locator = raw_ref.get("referenceLocator") + if reference_type == "purl": + continue + if not isinstance(locator, str) or not locator.strip(): + continue + urls.append((locator, _spdx_reference_source(reference_type))) + + return tuple(urls) + + +def _spdx_reference_source(reference_type: object) -> str: + if not isinstance(reference_type, str): + return "spdx.externalRefs" + normalized = reference_type.strip().lower() + tokens = tuple(token for token in re.split(r"[^a-z]+", normalized) if token) + if any(token in {"vcs", "scm", "git"} for token in tokens): + return "spdx.externalRefs.vcs" + return "spdx.externalRefs" + + +def _normalize_repository_url(raw_url: str, *, source: str) -> RepositoryMapping | None: + url = raw_url.strip() + if not url: + return None + + normalized = url + while normalized.startswith("git+"): + normalized = normalized[4:] + + if normalized.startswith("git@"): + normalized = f"ssh://{normalized.replace(':', '/', 1)}" + + parsed = urlparse(normalized) + host = (parsed.hostname or "").strip().lower() + if host not in _SUPPORTED_SCORECARD_PLATFORMS: + return None + + path_segments = [segment for segment in parsed.path.split("/") if segment] + if len(path_segments) != 2: + return None + + owner = path_segments[0].strip() + repo = path_segments[1].strip() + if repo.endswith(".git"): + repo = repo[:-4] + if not owner or not repo: + return None + + canonical_name = f"{host}/{owner}/{repo}" + return RepositoryMapping( + platform=host, + owner=owner, + repo=repo, + canonical_name=canonical_name, + repository_url=f"https://{canonical_name}", + source=source, + confidence=_source_confidence(source), + ) + + +def _dedupe_candidates(candidates: list[RepositoryMappingCandidate]) -> list[RepositoryMappingCandidate]: + deduped: dict[tuple[str, str], RepositoryMappingCandidate] = {} + for candidate in candidates: + key = (candidate.mapping.canonical_name, candidate.mapping.source) + existing = deduped.get(key) + if existing is None or candidate.priority < existing.priority: + deduped[key] = candidate + return list(deduped.values()) + + +def _source_confidence(source: str) -> RepositoryMappingConfidence: + if source in _HIGH_CONFIDENCE_SOURCES: + return RepositoryMappingConfidence.HIGH + return RepositoryMappingConfidence.LOW diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/scorecard_client.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/scorecard_client.py new file mode 100644 index 0000000..de5809f --- /dev/null +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/scorecard_client.py @@ -0,0 +1,193 @@ +from __future__ import annotations + +import json +import socket +from dataclasses import dataclass +from typing import Any +from urllib import error, parse, request + +from .models import ScorecardCheck + + +@dataclass(slots=True, frozen=True) +class ScorecardProjectResult: + canonical_name: str + score: float + date: str | None + scorecard_version: str | None + scorecard_commit: str | None + repository_commit: str | None + checks: tuple[ScorecardCheck, ...] + + +class ScorecardClientError(RuntimeError): + def __init__( + self, + message: str, + *, + status_code: int | None = None, + is_timeout: bool = False, + ) -> None: + super().__init__(message) + self.status_code = status_code + self.is_timeout = is_timeout + + +class ScorecardClient: + def __init__( + self, + *, + timeout_seconds: float = 5.0, + base_url: str = "https://api.securityscorecards.dev", + opener: request.OpenerDirector | None = None, + ) -> None: + self.timeout_seconds = timeout_seconds + self.base_url = base_url.rstrip("/") + self._opener = opener or request.build_opener() + + def fetch_project(self, platform: str, owner: str, repo: str) -> ScorecardProjectResult: + encoded_platform = parse.quote(platform, safe="") + encoded_owner = parse.quote(owner, safe="") + encoded_repo = parse.quote(repo, safe="") + path = f"/projects/{encoded_platform}/{encoded_owner}/{encoded_repo}" + payload = self._read_json(path) + return parse_project_payload(payload, expected_canonical_name=f"{platform}/{owner}/{repo}") + + def _read_json(self, path: str) -> object: + url = f"{self.base_url}{path}" + req = request.Request( + url, + headers={ + "Accept": "application/json", + "User-Agent": "sbom-diff-and-risk scorecard-client", + }, + ) + try: + with self._opener.open(req, timeout=self.timeout_seconds) as response: + payload = json.load(response) + except error.HTTPError as exc: + raise ScorecardClientError( + f"Scorecard request failed with HTTP {exc.code} for {url}.", + status_code=exc.code, + ) from exc + except error.URLError as exc: + if _is_timeout_reason(exc.reason): + raise ScorecardClientError( + f"Scorecard request timed out after {self.timeout_seconds} seconds for {url}.", + is_timeout=True, + ) from exc + raise ScorecardClientError(f"Scorecard request failed for {url}: {exc.reason}.") from exc + except TimeoutError as exc: + raise ScorecardClientError( + f"Scorecard request timed out after {self.timeout_seconds} seconds for {url}.", + is_timeout=True, + ) from exc + except socket.timeout as exc: + raise ScorecardClientError( + f"Scorecard request timed out after {self.timeout_seconds} seconds for {url}.", + is_timeout=True, + ) from exc + except json.JSONDecodeError as exc: + raise ScorecardClientError( + f"Scorecard returned malformed JSON for {url}: line {exc.lineno}, column {exc.colno}: {exc.msg}." + ) from exc + + return payload + + +def parse_project_payload(payload: object, *, expected_canonical_name: str) -> ScorecardProjectResult: + if not isinstance(payload, dict): + raise ScorecardClientError("Scorecard response must be a JSON object.") + + raw_repo = payload.get("repo") + repo_name = expected_canonical_name + repo_commit = None + if raw_repo is not None: + if not isinstance(raw_repo, dict): + raise ScorecardClientError("Scorecard repo field must be an object when present.") + repo_name = _optional_text(raw_repo.get("name")) or expected_canonical_name + repo_commit = _optional_text(raw_repo.get("commit")) + + score = _required_number(payload.get("score"), "Scorecard score") + date = _optional_text(payload.get("date")) + + scorecard_version = None + scorecard_commit = None + raw_scorecard = payload.get("scorecard") + if raw_scorecard is not None: + if not isinstance(raw_scorecard, dict): + raise ScorecardClientError("Scorecard metadata field must be an object when present.") + scorecard_version = _optional_text(raw_scorecard.get("version")) + scorecard_commit = _optional_text(raw_scorecard.get("commit")) + + raw_checks = payload.get("checks") + if raw_checks is None: + raw_checks = [] + if not isinstance(raw_checks, list): + raise ScorecardClientError("Scorecard checks field must be a list.") + + checks: list[ScorecardCheck] = [] + for raw_check in raw_checks: + if not isinstance(raw_check, dict): + raise ScorecardClientError("Scorecard check entries must be JSON objects.") + name = _required_text(raw_check.get("name"), "Scorecard check name") + raw_check_score = raw_check.get("score") + if raw_check_score is None: + continue + checks.append( + ScorecardCheck( + name=name, + score=int(_required_number(raw_check_score, f"Scorecard check {name} score")), + reason=_optional_text(raw_check.get("reason")), + documentation_url=_documentation_field(raw_check.get("documentation"), "url"), + documentation_short=_documentation_field(raw_check.get("documentation"), "short"), + ) + ) + + checks.sort(key=lambda item: item.name.lower()) + return ScorecardProjectResult( + canonical_name=repo_name, + score=score, + date=date, + scorecard_version=scorecard_version, + scorecard_commit=scorecard_commit, + repository_commit=repo_commit, + checks=tuple(checks), + ) + + +def _documentation_field(value: object, field: str) -> str | None: + if value is None: + return None + if not isinstance(value, dict): + raise ScorecardClientError("Scorecard documentation field must be an object when present.") + return _optional_text(value.get(field)) + + +def _required_text(value: object, context: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise ScorecardClientError(f"{context} must be a non-empty string.") + return value.strip() + + +def _required_number(value: object, context: str) -> float: + if not isinstance(value, (int, float)): + raise ScorecardClientError(f"{context} must be a number.") + return float(value) + + +def _optional_text(value: object) -> str | None: + if value is None: + return None + if not isinstance(value, str): + raise ScorecardClientError("Expected a string value in Scorecard response.") + stripped = value.strip() + return stripped or None + + +def _is_timeout_reason(reason: object) -> bool: + if isinstance(reason, (TimeoutError, socket.timeout)): + return True + if isinstance(reason, str): + return "timed out" in reason.lower() + return False diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/scorecard_enrichment.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/scorecard_enrichment.py new file mode 100644 index 0000000..43a4984 --- /dev/null +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/scorecard_enrichment.py @@ -0,0 +1,211 @@ +from __future__ import annotations + +from collections import Counter +from dataclasses import dataclass, replace + +from .models import ( + Component, + ReportEnrichmentMetadata, + RepositoryMapping, + ScorecardCheck, + ScorecardEvidence, + ScorecardStatus, +) +from .repository_mapping import assess_component_repository_mapping, repository_mapping_cache_key +from .scorecard_client import ScorecardClient, ScorecardClientError, ScorecardProjectResult + +DEFAULT_SCORECARD_TIMEOUT_SECONDS = 5.0 + +_STATUS_ORDER = { + ScorecardStatus.SCORECARD_AVAILABLE: 0, + ScorecardStatus.SCORECARD_UNAVAILABLE: 1, + ScorecardStatus.REPOSITORY_UNMAPPED: 2, + ScorecardStatus.ENRICHMENT_ERROR: 3, +} + + +@dataclass(slots=True, frozen=True) +class _ScorecardFetchOutcome: + status: ScorecardStatus + result: ScorecardProjectResult | None = None + note: str | None = None + error: str | None = None + + +class ScorecardEnricher: + def __init__( + self, + *, + client: ScorecardClient | None = None, + timeout_seconds: float = DEFAULT_SCORECARD_TIMEOUT_SECONDS, + ) -> None: + self.client = client or ScorecardClient(timeout_seconds=timeout_seconds) + self.timeout_seconds = timeout_seconds + self._component_cache: dict[tuple[str, str, str, tuple[tuple[str, str], ...]], ScorecardEvidence] = {} + self._repo_cache: dict[str, _ScorecardFetchOutcome] = {} + self._seen_keys: set[tuple[str, str, str, tuple[tuple[str, str], ...]]] = set() + + def enrich_components(self, components: list[Component]) -> list[Component]: + enriched: list[Component] = [] + for component in components: + key = _component_identity(component) + if key not in self._component_cache: + self._seen_keys.add(key) + self._component_cache[key] = self._enrich_component(component) + enriched.append(replace(component, scorecard=self._component_cache[key])) + return enriched + + def build_report_metadata(self) -> ReportEnrichmentMetadata: + if not self._seen_keys: + return ReportEnrichmentMetadata( + mode="opt_in_scorecard", + scorecard_enabled=True, + scorecard_timeout_seconds=self.timeout_seconds, + scorecard_network_access_performed=False, + network_access_performed=False, + scorecard_candidate_components=0, + scorecard_supported_components=0, + scorecard_status_counts={}, + ) + + evidences = [self._component_cache[key] for key in sorted(self._seen_keys)] + counter = Counter( + status.value + for evidence in evidences + for status in evidence.statuses + ) + scorecard_network_access_performed = any( + ScorecardStatus.REPOSITORY_UNMAPPED not in evidence.statuses + for evidence in evidences + ) + return ReportEnrichmentMetadata( + mode="opt_in_scorecard", + scorecard_enabled=True, + scorecard_timeout_seconds=self.timeout_seconds, + scorecard_network_access_performed=scorecard_network_access_performed, + network_access_performed=scorecard_network_access_performed, + scorecard_candidate_components=len(evidences), + scorecard_supported_components=sum( + 1 for evidence in evidences if ScorecardStatus.REPOSITORY_UNMAPPED not in evidence.statuses + ), + scorecard_status_counts={key: counter[key] for key in sorted(counter)}, + ) + + def _enrich_component(self, component: Component) -> ScorecardEvidence: + mapping_assessment = assess_component_repository_mapping(component) + mapping = mapping_assessment.mapping + if mapping is None: + return ScorecardEvidence( + provider="openssf-scorecard", + requested=True, + repository=None, + statuses=(ScorecardStatus.REPOSITORY_UNMAPPED,), + note="No high-confidence source repository mapping was available from explicit component metadata.", + ) + + outcome = self._repo_cache.get(mapping.canonical_name) + if outcome is None: + outcome = self._fetch_scorecard(mapping) + self._repo_cache[mapping.canonical_name] = outcome + return _scorecard_evidence_from_outcome(mapping, outcome) + + def _fetch_scorecard(self, mapping: RepositoryMapping) -> _ScorecardFetchOutcome: + try: + result = self.client.fetch_project(mapping.platform, mapping.owner, mapping.repo) + except ScorecardClientError as exc: + if exc.status_code == 404: + return _ScorecardFetchOutcome( + status=ScorecardStatus.SCORECARD_UNAVAILABLE, + note="Scorecard data is not available for the mapped repository.", + ) + return _ScorecardFetchOutcome( + status=ScorecardStatus.ENRICHMENT_ERROR, + error=str(exc), + ) + + return _ScorecardFetchOutcome( + status=ScorecardStatus.SCORECARD_AVAILABLE, + result=result, + ) + + +def scorecard_evidence_to_dict(evidence: ScorecardEvidence | None) -> dict[str, object] | None: + if evidence is None: + return None + repository = None + if evidence.repository is not None: + repository = { + "platform": evidence.repository.platform, + "owner": evidence.repository.owner, + "repo": evidence.repository.repo, + "canonical_name": evidence.repository.canonical_name, + "repository_url": evidence.repository.repository_url, + "source": evidence.repository.source, + "confidence": evidence.repository.confidence.value, + } + return { + "provider": evidence.provider, + "requested": evidence.requested, + "repository": repository, + "statuses": [status.value for status in evidence.statuses], + "score": evidence.score, + "date": evidence.date, + "scorecard_version": evidence.scorecard_version, + "scorecard_commit": evidence.scorecard_commit, + "repository_commit": evidence.repository_commit, + "checks": [ + { + "name": check.name, + "score": check.score, + "reason": check.reason, + "documentation_url": check.documentation_url, + "documentation_short": check.documentation_short, + } + for check in evidence.checks + ], + "note": evidence.note, + "error": evidence.error, + } + + +def _scorecard_evidence_from_outcome( + mapping: RepositoryMapping, + outcome: _ScorecardFetchOutcome, +) -> ScorecardEvidence: + if outcome.status is ScorecardStatus.SCORECARD_AVAILABLE: + assert outcome.result is not None + return ScorecardEvidence( + provider="openssf-scorecard", + requested=True, + repository=mapping, + statuses=(ScorecardStatus.SCORECARD_AVAILABLE,), + score=outcome.result.score, + date=outcome.result.date, + scorecard_version=outcome.result.scorecard_version, + scorecard_commit=outcome.result.scorecard_commit, + repository_commit=outcome.result.repository_commit, + checks=tuple( + sorted( + outcome.result.checks, + key=lambda item: (item.score, item.name.lower()), + ) + ), + ) + + return ScorecardEvidence( + provider="openssf-scorecard", + requested=True, + repository=mapping, + statuses=(outcome.status,), + checks=(), + note=outcome.note, + error=outcome.error, + ) + + +def _component_identity(component: Component) -> tuple[str, str, str]: + return repository_mapping_cache_key(component) + + +def _sorted_statuses(statuses: set[ScorecardStatus]) -> tuple[ScorecardStatus, ...]: + return tuple(sorted(statuses, key=lambda item: (_STATUS_ORDER[item], item.value))) diff --git a/tools/sbom-diff-and-risk/tests/test_cli_exit_codes.py b/tools/sbom-diff-and-risk/tests/test_cli_exit_codes.py index d184209..5a8adc6 100644 --- a/tools/sbom-diff-and-risk/tests/test_cli_exit_codes.py +++ b/tools/sbom-diff-and-risk/tests/test_cli_exit_codes.py @@ -123,6 +123,10 @@ def test_cli_compare_help_mentions_policy_flags_and_exit_codes() -> None: assert "--fail-on" in result.stdout assert "--warn-on" in result.stdout assert "--strict" in result.stdout + assert "--enrich-pypi" in result.stdout + assert "--pypi-timeout" in result.stdout + assert "--enrich-scorecard" in result.stdout + assert "--scorecard-timeout" in result.stdout assert "Exit codes: 0 = success/no blocking violations" in result.stdout diff --git a/tools/sbom-diff-and-risk/tests/test_cli_no_enrichment_regression.py b/tools/sbom-diff-and-risk/tests/test_cli_no_enrichment_regression.py new file mode 100644 index 0000000..6fdc27c --- /dev/null +++ b/tools/sbom-diff-and-risk/tests/test_cli_no_enrichment_regression.py @@ -0,0 +1,199 @@ +from __future__ import annotations + +import json +from pathlib import Path + +from sbom_diff_risk import cli +from sbom_diff_risk.models import ReportEnrichmentMetadata + + +def test_compare_stays_offline_and_deterministic_without_enrichment_flags( + monkeypatch, + tmp_path: Path, +) -> None: + project_root = Path(__file__).resolve().parents[1] + before = project_root / "examples" / "requirements_before.txt" + after = project_root / "examples" / "requirements_after.txt" + first_out = tmp_path / "first.json" + second_out = tmp_path / "second.json" + + class UnexpectedPyPIEnricher: + def __init__(self, *args, **kwargs) -> None: # noqa: ANN002, ANN003 + raise AssertionError("PyPI enrichment should remain disabled unless --enrich-pypi is set.") + + class UnexpectedScorecardEnricher: + def __init__(self, *args, **kwargs) -> None: # noqa: ANN002, ANN003 + raise AssertionError("Scorecard enrichment should remain disabled unless --enrich-scorecard is set.") + + monkeypatch.setattr(cli, "PyPIProvenanceEnricher", UnexpectedPyPIEnricher) + monkeypatch.setattr(cli, "ScorecardEnricher", UnexpectedScorecardEnricher) + + first_exit = cli.main( + [ + "compare", + "--before", + str(before), + "--after", + str(after), + "--out-json", + str(first_out), + ] + ) + second_exit = cli.main( + [ + "compare", + "--before", + str(before), + "--after", + str(after), + "--out-json", + str(second_out), + ] + ) + + assert first_exit == 0 + assert second_exit == 0 + assert first_out.read_text(encoding="utf-8") == second_out.read_text(encoding="utf-8") + + payload = json.loads(first_out.read_text(encoding="utf-8")) + assert payload["metadata"]["enrichment"]["mode"] == "offline_default" + assert payload["metadata"]["enrichment"]["pypi_enabled"] is False + assert payload["metadata"]["enrichment"]["network_access_performed"] is False + assert payload["trust_signal_notes"] == [ + "PyPI components are present, but provenance enrichment was not enabled for this run." + ] + + +def test_compare_runs_pypi_enrichment_only_when_requested( + monkeypatch, + tmp_path: Path, +) -> None: + project_root = Path(__file__).resolve().parents[1] + before = project_root / "examples" / "requirements_before.txt" + after = project_root / "examples" / "requirements_after.txt" + out_json = tmp_path / "enriched.json" + + class RecordingPyPIEnricher: + instances: list["RecordingPyPIEnricher"] = [] + + def __init__(self, *args, timeout_seconds: float, **kwargs) -> None: # noqa: ANN002, ANN003 + self.timeout_seconds = timeout_seconds + self.enrich_calls = 0 + self.__class__.instances.append(self) + + def enrich_components(self, components): # noqa: ANN001 + self.enrich_calls += 1 + return components + + def build_report_metadata(self) -> ReportEnrichmentMetadata: + return ReportEnrichmentMetadata( + mode="opt_in_pypi", + pypi_enabled=True, + pypi_timeout_seconds=self.timeout_seconds, + pypi_network_access_performed=False, + network_access_performed=False, + candidate_components=2, + supported_components=2, + status_counts={"attestation_unavailable": 2}, + ) + + class UnexpectedScorecardEnricher: + def __init__(self, *args, **kwargs) -> None: # noqa: ANN002, ANN003 + raise AssertionError("Scorecard enrichment should remain disabled unless --enrich-scorecard is set.") + + monkeypatch.setattr(cli, "PyPIProvenanceEnricher", RecordingPyPIEnricher) + monkeypatch.setattr(cli, "ScorecardEnricher", UnexpectedScorecardEnricher) + + exit_code = cli.main( + [ + "compare", + "--before", + str(before), + "--after", + str(after), + "--enrich-pypi", + "--pypi-timeout", + "2.5", + "--out-json", + str(out_json), + ] + ) + + payload = json.loads(out_json.read_text(encoding="utf-8")) + + assert exit_code == 0 + assert len(RecordingPyPIEnricher.instances) == 1 + assert RecordingPyPIEnricher.instances[0].timeout_seconds == 2.5 + assert RecordingPyPIEnricher.instances[0].enrich_calls == 2 + assert payload["metadata"]["enrichment"]["mode"] == "opt_in_pypi" + assert payload["metadata"]["enrichment"]["pypi_enabled"] is True + assert payload["metadata"]["enrichment"]["pypi_timeout_seconds"] == 2.5 + assert payload["notes"][1] == "PyPI provenance enrichment was requested explicitly." + assert payload["trust_signal_notes"] == [] + + +def test_compare_runs_scorecard_enrichment_only_when_requested( + monkeypatch, + tmp_path: Path, +) -> None: + project_root = Path(__file__).resolve().parents[1] + before = project_root / "examples" / "requirements_before.txt" + after = project_root / "examples" / "requirements_after.txt" + out_json = tmp_path / "scorecard.json" + + class UnexpectedPyPIEnricher: + def __init__(self, *args, **kwargs) -> None: # noqa: ANN002, ANN003 + raise AssertionError("PyPI enrichment should remain disabled unless --enrich-pypi is set.") + + class RecordingScorecardEnricher: + instances: list["RecordingScorecardEnricher"] = [] + + def __init__(self, *args, timeout_seconds: float, **kwargs) -> None: # noqa: ANN002, ANN003 + self.timeout_seconds = timeout_seconds + self.enrich_calls = 0 + self.__class__.instances.append(self) + + def enrich_components(self, components): # noqa: ANN001 + self.enrich_calls += 1 + return components + + def build_report_metadata(self) -> ReportEnrichmentMetadata: + return ReportEnrichmentMetadata( + mode="opt_in_scorecard", + scorecard_enabled=True, + scorecard_timeout_seconds=self.timeout_seconds, + scorecard_network_access_performed=False, + network_access_performed=False, + scorecard_candidate_components=2, + scorecard_supported_components=1, + scorecard_status_counts={"repository_unmapped": 1}, + ) + + monkeypatch.setattr(cli, "PyPIProvenanceEnricher", UnexpectedPyPIEnricher) + monkeypatch.setattr(cli, "ScorecardEnricher", RecordingScorecardEnricher) + + exit_code = cli.main( + [ + "compare", + "--before", + str(before), + "--after", + str(after), + "--enrich-scorecard", + "--scorecard-timeout", + "4.25", + "--out-json", + str(out_json), + ] + ) + + payload = json.loads(out_json.read_text(encoding="utf-8")) + + assert exit_code == 0 + assert len(RecordingScorecardEnricher.instances) == 1 + assert RecordingScorecardEnricher.instances[0].timeout_seconds == 4.25 + assert RecordingScorecardEnricher.instances[0].enrich_calls == 2 + assert payload["metadata"]["enrichment"]["mode"] == "opt_in_scorecard" + assert payload["metadata"]["enrichment"]["scorecard_enabled"] is True + assert payload["metadata"]["enrichment"]["scorecard_timeout_seconds"] == 4.25 + assert payload["notes"][1] == "OpenSSF Scorecard enrichment was requested explicitly." diff --git a/tools/sbom-diff-and-risk/tests/test_cli_provenance_policy.py b/tools/sbom-diff-and-risk/tests/test_cli_provenance_policy.py new file mode 100644 index 0000000..59e2d3c --- /dev/null +++ b/tools/sbom-diff-and-risk/tests/test_cli_provenance_policy.py @@ -0,0 +1,311 @@ +from __future__ import annotations + +import argparse +import json +from pathlib import Path + +from sbom_diff_risk.cli import run_compare +from sbom_diff_risk.models import ProvenanceEvidence, ProvenanceFileEvidence, ProvenanceStatus, ReportEnrichmentMetadata + + +def test_run_compare_blocks_on_provenance_required_with_mocked_enrichment( + tmp_path: Path, + monkeypatch, +) -> None: + project_root = Path(__file__).resolve().parents[1] + policy_path = tmp_path / "policy.yml" + policy_path.write_text( + "\n".join( + [ + "version: 2", + "block_on:", + " - provenance_required", + "require_attestations_for_new_packages: true", + "", + ] + ), + encoding="utf-8", + ) + monkeypatch.setattr("sbom_diff_risk.cli.PyPIProvenanceEnricher", FakeMissingAttestationEnricher) + + exit_code = run_compare( + argparse.Namespace( + before=project_root / "examples" / "requirements_before.txt", + after=project_root / "examples" / "requirements_after.txt", + format="auto", + before_format=None, + after_format=None, + pyproject_group=None, + out_json=tmp_path / "report.json", + out_md=None, + out_sarif=None, + policy=policy_path, + fail_on=None, + warn_on=None, + strict=False, + enrich_pypi=True, + pypi_timeout=2.0, + source_allowlist="pypi.org,files.pythonhosted.org,github.com", + ) + ) + + payload = json.loads((tmp_path / "report.json").read_text(encoding="utf-8")) + + assert exit_code == 1 + assert any(item["rule_id"] == "provenance_required" for item in payload["blocking_findings"]) + + +def test_run_compare_passes_when_package_is_allowlisted_for_missing_attestation( + tmp_path: Path, + monkeypatch, +) -> None: + project_root = Path(__file__).resolve().parents[1] + policy_path = tmp_path / "policy.yml" + policy_path.write_text( + "\n".join( + [ + "version: 2", + "block_on:", + " - provenance_required", + "require_attestations_for_new_packages: true", + "allow_unattested_packages:", + " - urllib3", + "", + ] + ), + encoding="utf-8", + ) + monkeypatch.setattr("sbom_diff_risk.cli.PyPIProvenanceEnricher", FakeMissingAttestationEnricher) + + exit_code = run_compare( + argparse.Namespace( + before=project_root / "examples" / "requirements_before.txt", + after=project_root / "examples" / "requirements_after.txt", + format="auto", + before_format=None, + after_format=None, + pyproject_group=None, + out_json=tmp_path / "report.json", + out_md=None, + out_sarif=None, + policy=policy_path, + fail_on=None, + warn_on=None, + strict=False, + enrich_pypi=True, + pypi_timeout=2.0, + source_allowlist="pypi.org,files.pythonhosted.org,github.com", + ) + ) + + payload = json.loads((tmp_path / "report.json").read_text(encoding="utf-8")) + + assert exit_code == 0 + assert payload["blocking_findings"] == [] + + +def test_run_compare_blocks_on_unverified_provenance_with_publisher_override_alias( + tmp_path: Path, + monkeypatch, +) -> None: + project_root = Path(__file__).resolve().parents[1] + policy_path = tmp_path / "policy.yml" + policy_path.write_text( + "\n".join( + [ + "version: 2", + "block_on:", + " - unverified_provenance", + "allow_unattested_publishers:", + " - github actions", + "", + ] + ), + encoding="utf-8", + ) + monkeypatch.setattr("sbom_diff_risk.cli.PyPIProvenanceEnricher", FakeUnverifiedPublisherEnricher) + + exit_code = run_compare( + argparse.Namespace( + before=project_root / "examples" / "requirements_before.txt", + after=project_root / "examples" / "requirements_after.txt", + format="auto", + before_format=None, + after_format=None, + pyproject_group=None, + out_json=tmp_path / "report.json", + out_md=None, + out_sarif=None, + policy=policy_path, + fail_on=None, + warn_on=None, + strict=False, + enrich_pypi=True, + pypi_timeout=2.0, + source_allowlist="pypi.org,files.pythonhosted.org,github.com", + ) + ) + + payload = json.loads((tmp_path / "report.json").read_text(encoding="utf-8")) + + assert exit_code == 1 + assert any(item["rule_id"] == "unverified_provenance" for item in payload["blocking_findings"]) + + +def test_run_compare_keeps_provenance_unavailable_blocking_even_for_allowlisted_package( + tmp_path: Path, + monkeypatch, +) -> None: + project_root = Path(__file__).resolve().parents[1] + policy_path = tmp_path / "policy.yml" + policy_path.write_text( + "\n".join( + [ + "version: 2", + "block_on:", + " - provenance_unavailable", + "allow_unattested_packages:", + " - urllib3", + "", + ] + ), + encoding="utf-8", + ) + monkeypatch.setattr("sbom_diff_risk.cli.PyPIProvenanceEnricher", FakeUnavailableProvenanceEnricher) + + exit_code = run_compare( + argparse.Namespace( + before=project_root / "examples" / "requirements_before.txt", + after=project_root / "examples" / "requirements_after.txt", + format="auto", + before_format=None, + after_format=None, + pyproject_group=None, + out_json=tmp_path / "report.json", + out_md=None, + out_sarif=None, + policy=policy_path, + fail_on=None, + warn_on=None, + strict=False, + enrich_pypi=True, + pypi_timeout=2.0, + source_allowlist="pypi.org,files.pythonhosted.org,github.com", + ) + ) + + payload = json.loads((tmp_path / "report.json").read_text(encoding="utf-8")) + + assert exit_code == 1 + assert any(item["rule_id"] == "provenance_unavailable" for item in payload["blocking_findings"]) + + +class FakeMissingAttestationEnricher: + def __init__(self, *, timeout_seconds: float) -> None: + self.timeout_seconds = timeout_seconds + + def enrich_components(self, components): # noqa: ANN001 + for component in components: + if component.ecosystem.strip().lower() != "pypi": + continue + component.provenance = ProvenanceEvidence( + provider="pypi", + requested=True, + package_name=component.name, + package_version=component.version, + release_url=f"https://pypi.org/project/{component.name}/{component.version}/", + statuses=(ProvenanceStatus.ATTESTATION_UNAVAILABLE,), + files=( + ProvenanceFileEvidence( + filename=f"{component.name}-{component.version}.tar.gz", + statuses=(ProvenanceStatus.ATTESTATION_UNAVAILABLE,), + attestation_count=0, + ), + ), + ) + return components + + def build_report_metadata(self) -> ReportEnrichmentMetadata: + return ReportEnrichmentMetadata( + mode="opt_in_pypi", + pypi_enabled=True, + pypi_timeout_seconds=self.timeout_seconds, + network_access_performed=True, + candidate_components=3, + supported_components=3, + status_counts={"attestation_unavailable": 3}, + ) + + +class FakeUnverifiedPublisherEnricher: + def __init__(self, *, timeout_seconds: float) -> None: + self.timeout_seconds = timeout_seconds + + def enrich_components(self, components): # noqa: ANN001 + for component in components: + if component.ecosystem.strip().lower() != "pypi": + continue + component.provenance = ProvenanceEvidence( + provider="pypi", + requested=True, + package_name=component.name, + package_version=component.version, + release_url=f"https://pypi.org/project/{component.name}/{component.version}/", + statuses=(ProvenanceStatus.PROVENANCE_AVAILABLE, ProvenanceStatus.ATTESTATION_AVAILABLE), + files=( + ProvenanceFileEvidence( + filename=f"{component.name}-{component.version}.tar.gz", + statuses=(ProvenanceStatus.PROVENANCE_AVAILABLE, ProvenanceStatus.ATTESTATION_AVAILABLE), + attestation_count=1, + publisher_kinds=("manual upload",), + ), + ), + ) + return components + + def build_report_metadata(self) -> ReportEnrichmentMetadata: + return ReportEnrichmentMetadata( + mode="opt_in_pypi", + pypi_enabled=True, + pypi_timeout_seconds=self.timeout_seconds, + pypi_network_access_performed=True, + network_access_performed=True, + candidate_components=3, + supported_components=3, + status_counts={ + "attestation_available": 3, + "provenance_available": 3, + }, + ) + + +class FakeUnavailableProvenanceEnricher: + def __init__(self, *, timeout_seconds: float) -> None: + self.timeout_seconds = timeout_seconds + + def enrich_components(self, components): # noqa: ANN001 + for component in components: + if component.ecosystem.strip().lower() != "pypi": + continue + component.provenance = ProvenanceEvidence( + provider="pypi", + requested=True, + package_name=component.name, + package_version=component.version, + release_url=f"https://pypi.org/project/{component.name}/{component.version}/", + statuses=(ProvenanceStatus.ENRICHMENT_ERROR,), + error="PyPI provenance evidence could not be fetched due to an enrichment error.", + ) + return components + + def build_report_metadata(self) -> ReportEnrichmentMetadata: + return ReportEnrichmentMetadata( + mode="opt_in_pypi", + pypi_enabled=True, + pypi_timeout_seconds=self.timeout_seconds, + pypi_network_access_performed=True, + network_access_performed=True, + candidate_components=3, + supported_components=3, + status_counts={"enrichment_error": 3}, + ) diff --git a/tools/sbom-diff-and-risk/tests/test_policy.py b/tools/sbom-diff-and-risk/tests/test_policy.py index 67f6f51..b64abb7 100644 --- a/tools/sbom-diff-and-risk/tests/test_policy.py +++ b/tools/sbom-diff-and-risk/tests/test_policy.py @@ -39,9 +39,9 @@ def test_policy_parser_rejects_unknown_key(tmp_path: Path) -> None: def test_policy_parser_rejects_invalid_version(tmp_path: Path) -> None: path = tmp_path / "policy.yml" - path.write_text("version: 2\n", encoding="utf-8") + path.write_text("version: 4\n", encoding="utf-8") - with pytest.raises(PolicyError, match="only version 1"): + with pytest.raises(PolicyError, match="versions 1, 2, and 3"): load_policy(path) diff --git a/tools/sbom-diff-and-risk/tests/test_policy_provenance.py b/tools/sbom-diff-and-risk/tests/test_policy_provenance.py new file mode 100644 index 0000000..1976942 --- /dev/null +++ b/tools/sbom-diff-and-risk/tests/test_policy_provenance.py @@ -0,0 +1,229 @@ +from __future__ import annotations + +from pathlib import Path + +import pytest + +from sbom_diff_risk.errors import PolicyError +from sbom_diff_risk.models import ( + Component, + ProvenanceEvidence, + ProvenanceFileEvidence, + ProvenanceStatus, + RiskBucket, + RiskFinding, +) +from sbom_diff_risk.policy_evaluator import evaluate_policy +from sbom_diff_risk.policy_models import PolicyConfig, PolicyLevel +from sbom_diff_risk.policy_parser import build_policy, load_policy + + +def test_policy_parser_accepts_provenance_v2_policy() -> None: + policy = load_policy(_example_path("policy-provenance-minimal.yml")) + + assert policy.version == 2 + assert policy.warn_on == ("missing_attestation", "provenance_required") + assert policy.require_attestations_for_new_packages is True + assert policy.allow_unattested_packages == ("pip",) + + +def test_policy_parser_rejects_v2_keys_in_version_1_policy(tmp_path: Path) -> None: + path = tmp_path / "policy.yml" + path.write_text("version: 1\nrequire_attestations_for_new_packages: true\n", encoding="utf-8") + + with pytest.raises(PolicyError, match="version 1 does not support keys"): + load_policy(path) + + +def test_policy_parser_rejects_v2_rule_ids_in_version_1_policy(tmp_path: Path) -> None: + path = tmp_path / "policy.yml" + path.write_text("version: 1\nblock_on: [missing_attestation]\n", encoding="utf-8") + + with pytest.raises(PolicyError, match="Unknown rule id"): + load_policy(path) + + +def test_policy_parser_accepts_allow_unattested_publishers_alias() -> None: + policy = load_policy(_example_path("policy-provenance-strict.yml")) + + assert policy.allow_provenance_publishers == ("github actions",) + + +def test_policy_parser_rejects_conflicting_publisher_override_keys(tmp_path: Path) -> None: + path = tmp_path / "policy.yml" + path.write_text( + "\n".join( + [ + "version: 2", + "allow_provenance_publishers:", + " - github actions", + "allow_unattested_publishers:", + " - manual upload", + "", + ] + ), + encoding="utf-8", + ) + + with pytest.raises(PolicyError, match="use either allow_provenance_publishers or allow_unattested_publishers"): + load_policy(path) + + +def test_build_policy_cli_only_provenance_rule_upgrades_to_version_2() -> None: + policy, policy_path = build_policy(fail_on="missing_attestation") + + assert policy_path is None + assert policy is not None + assert policy.version == 2 + assert policy.block_on == ("missing_attestation",) + + +def test_policy_evaluator_warns_on_provenance_unavailable_without_enrichment() -> None: + policy = PolicyConfig(version=2, warn_on=("provenance_unavailable",)) + component = Component(name="urllib3", version="2.2.1", ecosystem="pypi") + + evaluation = evaluate_policy(policy, policy_path="policy.yml", added=[component], changed=[], findings=[]) + + assert evaluation.exit_code == 0 + assert len(evaluation.warning_violations) == 1 + assert evaluation.warning_violations[0].rule_id == "provenance_unavailable" + + +def test_policy_evaluator_blocks_on_missing_attestation_when_release_is_unattested() -> None: + policy = PolicyConfig(version=2, block_on=("missing_attestation",)) + component = _component_with_provenance( + "urllib3", + "2.2.1", + statuses=(ProvenanceStatus.ATTESTATION_UNAVAILABLE,), + ) + + evaluation = evaluate_policy(policy, policy_path="policy.yml", added=[component], changed=[], findings=[]) + + assert evaluation.exit_code == 1 + assert evaluation.blocking_violations[0].rule_id == "missing_attestation" + assert evaluation.blocking_violations[0].level is PolicyLevel.BLOCK + + +def test_policy_evaluator_blocks_on_unverified_provenance_when_publishers_do_not_match() -> None: + policy = PolicyConfig( + version=2, + block_on=("unverified_provenance",), + allow_provenance_publishers=("github actions",), + ) + component = _component_with_provenance( + "requests", + "2.32.0", + statuses=(ProvenanceStatus.PROVENANCE_AVAILABLE, ProvenanceStatus.ATTESTATION_AVAILABLE), + publisher_kinds=("manual upload",), + ) + + evaluation = evaluate_policy(policy, policy_path="policy.yml", added=[component], changed=[], findings=[]) + + assert evaluation.exit_code == 1 + assert evaluation.blocking_violations[0].rule_id == "unverified_provenance" + assert "allow_provenance_publishers" in evaluation.blocking_violations[0].message + + +def test_policy_evaluator_allows_explicit_unattested_package_override() -> None: + policy = PolicyConfig( + version=2, + require_attestations_for_new_packages=True, + allow_unattested_packages=("urllib3",), + ) + component = _component_with_provenance( + "urllib3", + "2.2.1", + statuses=(ProvenanceStatus.ATTESTATION_UNAVAILABLE,), + ) + + evaluation = evaluate_policy(policy, policy_path="policy.yml", added=[component], changed=[], findings=[]) + + assert evaluation.exit_code == 0 + assert evaluation.blocking_violations == [] + assert evaluation.warning_violations == [] + + +def test_policy_evaluator_allow_unattested_package_does_not_suppress_provenance_unavailable() -> None: + policy = PolicyConfig( + version=2, + block_on=("provenance_unavailable",), + allow_unattested_packages=("urllib3",), + ) + component = Component(name="urllib3", version="2.2.1", ecosystem="pypi") + + evaluation = evaluate_policy(policy, policy_path="policy.yml", added=[component], changed=[], findings=[]) + + assert evaluation.exit_code == 1 + assert [violation.rule_id for violation in evaluation.blocking_violations] == ["provenance_unavailable"] + + +def test_policy_evaluator_keeps_v1_behavior_when_enrichment_evidence_is_present() -> None: + policy = PolicyConfig(version=1, warn_on=("new_package",)) + component = _component_with_provenance( + "urllib3", + "2.2.1", + statuses=(ProvenanceStatus.ATTESTATION_UNAVAILABLE,), + ) + finding = RiskFinding( + bucket=RiskBucket.NEW_PACKAGE, + component_key="coord:pypi:urllib3", + component=component, + rationale="Component was not present in the before input.", + ) + + evaluation = evaluate_policy(policy, policy_path="policy.yml", added=[component], changed=[], findings=[finding]) + + assert evaluation.exit_code == 0 + assert [violation.rule_id for violation in evaluation.warning_violations] == ["new_package"] + assert evaluation.blocking_violations == [] + + +def test_policy_evaluator_blocks_when_suspicious_source_requires_provenance() -> None: + policy = PolicyConfig(version=2, require_provenance_for_suspicious_sources=True) + component = Component(name="mystery-lib", version="1.0.0", ecosystem="pypi", source_url="http://example.test/mystery-lib") + finding = RiskFinding( + bucket=RiskBucket.SUSPICIOUS_SOURCE, + component_key="coord:pypi:mystery-lib", + component=component, + rationale="Source provenance is suspicious.", + ) + + evaluation = evaluate_policy(policy, policy_path="policy.yml", added=[component], changed=[], findings=[finding]) + + assert evaluation.exit_code == 1 + assert evaluation.blocking_violations[0].rule_id == "provenance_required" + assert "suspicious source" in evaluation.blocking_violations[0].message + + +def _component_with_provenance( + name: str, + version: str, + *, + statuses: tuple[ProvenanceStatus, ...], + publisher_kinds: tuple[str, ...] = (), +) -> Component: + return Component( + name=name, + version=version, + ecosystem="pypi", + provenance=ProvenanceEvidence( + provider="pypi", + requested=True, + package_name=name, + package_version=version, + release_url=f"https://pypi.org/project/{name}/{version}/", + statuses=statuses, + files=( + ProvenanceFileEvidence( + filename=f"{name}-{version}.tar.gz", + statuses=statuses, + attestation_count=1 if ProvenanceStatus.ATTESTATION_AVAILABLE in statuses else 0, + publisher_kinds=publisher_kinds, + ), + ), + ), + ) + + +def _example_path(name: str) -> Path: + return Path(__file__).resolve().parents[1] / "examples" / name diff --git a/tools/sbom-diff-and-risk/tests/test_policy_scorecard.py b/tools/sbom-diff-and-risk/tests/test_policy_scorecard.py new file mode 100644 index 0000000..290119c --- /dev/null +++ b/tools/sbom-diff-and-risk/tests/test_policy_scorecard.py @@ -0,0 +1,107 @@ +from __future__ import annotations + +import pytest + +from sbom_diff_risk.errors import PolicyError +from sbom_diff_risk.models import ( + Component, + RepositoryMapping, + ScorecardCheck, + ScorecardEvidence, + ScorecardStatus, +) +from sbom_diff_risk.policy_evaluator import evaluate_policy +from sbom_diff_risk.policy_models import PolicyConfig +from sbom_diff_risk.policy_parser import build_policy, load_policy + + +def test_policy_parser_accepts_scorecard_v3_policy(tmp_path) -> None: # noqa: ANN001 + path = tmp_path / "policy.yml" + path.write_text( + "\n".join( + [ + "version: 3", + "warn_on: [scorecard_below_threshold]", + "minimum_scorecard_score: 7.5", + "", + ] + ), + encoding="utf-8", + ) + + policy = load_policy(path) + + assert policy.version == 3 + assert policy.warn_on == ("scorecard_below_threshold",) + assert policy.minimum_scorecard_score == 7.5 + + +def test_policy_parser_rejects_scorecard_keys_in_version_2_policy(tmp_path: Path) -> None: + path = tmp_path / "policy.yml" + path.write_text("version: 2\nminimum_scorecard_score: 7.0\n", encoding="utf-8") + + with pytest.raises(PolicyError, match="version 2 does not support keys"): + load_policy(path) + + +def test_build_policy_cli_only_scorecard_rule_upgrades_to_version_3() -> None: + policy, policy_path = build_policy(fail_on="scorecard_below_threshold") + + assert policy_path is None + assert policy is not None + assert policy.version == 3 + assert policy.block_on == ("scorecard_below_threshold",) + + +def test_policy_evaluator_warns_when_scorecard_below_threshold() -> None: + policy = PolicyConfig( + version=3, + warn_on=("scorecard_below_threshold",), + minimum_scorecard_score=7.0, + ) + component = _component_with_scorecard("requests", "2.32.0", score=5.5) + + evaluation = evaluate_policy(policy, policy_path="policy.yml", added=[component], changed=[], findings=[]) + + assert evaluation.exit_code == 0 + assert len(evaluation.warning_violations) == 1 + assert evaluation.warning_violations[0].rule_id == "scorecard_below_threshold" + assert "minimum_scorecard_score=7.0" in evaluation.warning_violations[0].message + + +def test_policy_evaluator_does_not_gate_scorecard_threshold_without_explicit_rule() -> None: + policy = PolicyConfig( + version=3, + minimum_scorecard_score=7.0, + ) + component = _component_with_scorecard("requests", "2.32.0", score=5.5) + + evaluation = evaluate_policy(policy, policy_path="policy.yml", added=[component], changed=[], findings=[]) + + assert evaluation.exit_code == 0 + assert evaluation.blocking_violations == [] + assert evaluation.warning_violations == [] + + +def _component_with_scorecard(name: str, version: str, *, score: float) -> Component: + return Component( + name=name, + version=version, + ecosystem="pypi", + scorecard=ScorecardEvidence( + provider="openssf-scorecard", + requested=True, + repository=RepositoryMapping( + platform="github.com", + owner="psf", + repo=name, + canonical_name=f"github.com/psf/{name}", + repository_url=f"https://github.com/psf/{name}", + source="component.source_url", + ), + statuses=(ScorecardStatus.SCORECARD_AVAILABLE,), + score=score, + date="2026-04-10T00:00:00Z", + checks=(ScorecardCheck(name="Maintained", score=10),), + ), + ) diff --git a/tools/sbom-diff-and-risk/tests/test_provenance_reporting.py b/tools/sbom-diff-and-risk/tests/test_provenance_reporting.py new file mode 100644 index 0000000..20321b7 --- /dev/null +++ b/tools/sbom-diff-and-risk/tests/test_provenance_reporting.py @@ -0,0 +1,456 @@ +from __future__ import annotations + +import json +from pathlib import Path + +from sbom_diff_risk.diffing import component_key +from sbom_diff_risk.models import ( + CompareReport, + Component, + ComponentChange, + ProvenanceEvidence, + ProvenanceFileEvidence, + ProvenanceStatus, + ReportComponents, + ReportEnrichmentMetadata, + ReportMetadata, + ReportSummary, + RiskBucket, + RiskFinding, +) +from sbom_diff_risk.policy_evaluator import evaluate_policy +from sbom_diff_risk.policy_models import PolicyConfig +from sbom_diff_risk.report_json import render_report_json +from sbom_diff_risk.report_md import render_report_markdown +from sbom_diff_risk.report_sarif import ( + render_report_sarif, + sarif_rule_id_for_policy_violation, +) +from sbom_diff_risk.risk import summarize_risks + + +def test_provenance_report_json_matches_golden() -> None: + report, _, _ = _build_sample_provenance_report() + + rendered = render_report_json(report) + expected = _read_example("sample-provenance-report.json") + + assert rendered == expected + + +def test_provenance_report_markdown_matches_golden() -> None: + report, _, _ = _build_sample_provenance_report() + + rendered = render_report_markdown(report) + expected = _read_example("sample-provenance-report.md") + + assert rendered == expected + + +def test_provenance_report_json_includes_provenance_policy_summary() -> None: + report, _, _ = _build_sample_provenance_report() + + payload = json.loads(render_report_json(report)) + + assert payload["provenance_policy"]["configured"] is True + assert payload["provenance_policy_impact"] == payload["provenance_policy"] + assert payload["provenance_policy"]["requirements"]["require_attestations_for_new_packages"] is True + assert payload["provenance_policy"]["requirements"]["allow_provenance_publishers"] == ["github actions"] + assert payload["provenance_policy"]["counts"] == {"blocking": 2, "warning": 1, "suppressed": 0} + assert payload["enrichment_metadata"]["pypi_network_access_performed"] is True + + +def test_provenance_component_evidence_includes_lookup_and_file_totals() -> None: + report, _, _ = _build_sample_provenance_report() + + payload = json.loads(render_report_json(report)) + urllib3_provenance = payload["components"]["added"][0]["evidence"]["provenance"] + mystery_lib_provenance = payload["components"]["added"][1]["evidence"]["provenance"] + + assert urllib3_provenance["supported"] is True + assert urllib3_provenance["lookup_performed"] is True + assert urllib3_provenance["files_evaluated"] == 1 + assert urllib3_provenance["files_with_attestations"] == 1 + assert urllib3_provenance["files_without_attestations"] == 0 + assert mystery_lib_provenance["supported"] is True + assert mystery_lib_provenance["lookup_performed"] is True + assert mystery_lib_provenance["files_evaluated"] == 1 + assert mystery_lib_provenance["files_with_attestations"] == 0 + assert mystery_lib_provenance["files_without_attestations"] == 1 + + +def test_blocking_missing_attestation_emits_sarif_alert() -> None: + component = Component( + name="urllib3", + version="2.2.1", + ecosystem="pypi", + provenance=_provenance( + name="urllib3", + version="2.2.1", + statuses=(ProvenanceStatus.ATTESTATION_UNAVAILABLE,), + file_statuses=(ProvenanceStatus.ATTESTATION_UNAVAILABLE,), + attestation_count=0, + ), + ) + policy = PolicyConfig(version=2, block_on=("missing_attestation",)) + report = CompareReport( + summary=ReportSummary(added=1, removed=0, changed=0, risk_counts=summarize_risks([])), + components=ReportComponents(added=[component], removed=[], changed=[]), + risks=[], + metadata=ReportMetadata( + before_format="requirements-txt", + after_format="requirements-txt", + generated_at=None, + strict=False, + stub=False, + policy_evaluation=evaluate_policy(policy, policy_path="policy.yml", added=[component], changed=[], findings=[]), + enrichment=ReportEnrichmentMetadata( + mode="opt_in_pypi", + pypi_enabled=True, + pypi_timeout_seconds=3.0, + pypi_network_access_performed=True, + network_access_performed=True, + candidate_components=1, + supported_components=1, + status_counts={"attestation_unavailable": 1}, + ), + ), + notes=["PyPI provenance enrichment was requested explicitly."], + ) + project_root = Path(__file__).resolve().parents[1] + examples = project_root / "examples" + + payload = json.loads( + render_report_sarif(report, before_path=examples / "requirements_before.txt", after_path=examples / "requirements_after.txt", base_dir=project_root) + ) + + results = payload["runs"][0]["results"] + + assert [result["ruleId"] for result in results] == ["sdr.policy_violation.missing_attestation"] + assert results[0]["message"]["text"] == "urllib3: No PyPI attestations were published for this release." + assert results[0]["locations"][0]["physicalLocation"]["artifactLocation"]["uri"] == "examples/requirements_after.txt" + + +def test_provenance_sarif_matches_golden() -> None: + report, before_path, after_path = _build_sample_provenance_report() + project_root = Path(__file__).resolve().parents[1] + + rendered = render_report_sarif(report, before_path=before_path, after_path=after_path, base_dir=project_root) + expected = _read_example("sample-provenance-report.sarif") + + assert _normalize_sarif_snapshot(rendered) == _normalize_sarif_snapshot(expected) + + +def test_provenance_sarif_rule_ids_are_stable() -> None: + assert sarif_rule_id_for_policy_violation("provenance_required") == "sdr.policy_violation.provenance_required" + assert sarif_rule_id_for_policy_violation("missing_attestation") == "sdr.policy_violation.missing_attestation" + assert sarif_rule_id_for_policy_violation("unverified_provenance") == "sdr.policy_violation.unverified_provenance" + assert sarif_rule_id_for_policy_violation("provenance_unavailable") is None + + +def test_non_blocking_missing_attestation_does_not_automatically_become_sarif_alert() -> None: + component = Component( + name="urllib3", + version="2.2.1", + ecosystem="pypi", + provenance=_provenance( + name="urllib3", + version="2.2.1", + statuses=(ProvenanceStatus.ATTESTATION_UNAVAILABLE,), + file_statuses=(ProvenanceStatus.ATTESTATION_UNAVAILABLE,), + attestation_count=0, + ), + ) + policy = PolicyConfig(version=2, warn_on=("missing_attestation",)) + report = CompareReport( + summary=ReportSummary(added=1, removed=0, changed=0, risk_counts=summarize_risks([])), + components=ReportComponents(added=[component], removed=[], changed=[]), + risks=[], + metadata=ReportMetadata( + before_format="requirements-txt", + after_format="requirements-txt", + generated_at=None, + strict=False, + stub=False, + policy_evaluation=evaluate_policy(policy, policy_path="policy.yml", added=[component], changed=[], findings=[]), + enrichment=ReportEnrichmentMetadata( + mode="opt_in_pypi", + pypi_enabled=True, + pypi_timeout_seconds=3.0, + pypi_network_access_performed=True, + network_access_performed=True, + candidate_components=1, + supported_components=1, + status_counts={"attestation_unavailable": 1}, + ), + ), + notes=["PyPI provenance enrichment was requested explicitly."], + ) + project_root = Path(__file__).resolve().parents[1] + examples = project_root / "examples" + + payload = json.loads( + render_report_sarif(report, before_path=examples / "requirements_before.txt", after_path=examples / "requirements_after.txt", base_dir=project_root) + ) + + assert payload["runs"][0]["results"] == [] + + +def test_provenance_sarif_prefers_primary_policy_alert_when_requirement_already_blocks() -> None: + report, before_path, after_path = _build_sample_provenance_report() + project_root = Path(__file__).resolve().parents[1] + + payload = json.loads( + render_report_sarif(report, before_path=before_path, after_path=after_path, base_dir=project_root) + ) + + results = payload["runs"][0]["results"] + + assert [result["ruleId"] for result in results] == [ + "sdr.policy_violation.provenance_required", + "sdr.policy_violation.unverified_provenance", + ] + assert results[0]["message"]["text"] == "mystery-lib: Provenance required for new package; no attestations were published." + assert results[1]["message"]["text"] == "legacy-lib: PyPI attestation publisher could not be verified by policy." + assert all( + result["locations"][0]["physicalLocation"]["artifactLocation"]["uri"] == "examples/requirements_after.txt" + for result in results + ) + + +def test_provenance_required_for_suspicious_source_emits_concise_sarif_alert() -> None: + component = Component( + name="mystery-lib", + version="1.0.0", + ecosystem="pypi", + source_url="http://example.test/mystery-lib", + ) + finding = RiskFinding( + bucket=RiskBucket.SUSPICIOUS_SOURCE, + component_key="coord:pypi:mystery-lib", + component=component, + rationale="Source provenance is suspicious.", + ) + policy = PolicyConfig(version=2, require_provenance_for_suspicious_sources=True) + report = CompareReport( + summary=ReportSummary(added=1, removed=0, changed=0, risk_counts=summarize_risks([finding])), + components=ReportComponents(added=[component], removed=[], changed=[]), + risks=[finding], + metadata=ReportMetadata( + before_format="requirements-txt", + after_format="requirements-txt", + generated_at=None, + strict=False, + stub=False, + policy_evaluation=evaluate_policy(policy, policy_path="policy.yml", added=[component], changed=[], findings=[finding]), + enrichment=ReportEnrichmentMetadata( + mode="offline_default", + pypi_enabled=False, + pypi_network_access_performed=False, + network_access_performed=False, + ), + ), + notes=["No network enrichment was performed."], + ) + project_root = Path(__file__).resolve().parents[1] + examples = project_root / "examples" + + payload = json.loads( + render_report_sarif(report, before_path=examples / "requirements_before.txt", after_path=examples / "requirements_after.txt", base_dir=project_root) + ) + + results = payload["runs"][0]["results"] + + assert [result["ruleId"] for result in results] == ["sdr.policy_violation.provenance_required"] + assert results[0]["message"]["text"] == "mystery-lib: Provenance required for suspicious source; provenance evidence was unavailable." + assert results[0]["locations"][0]["physicalLocation"]["artifactLocation"]["uri"] == "examples/requirements_after.txt" + + +def _build_sample_provenance_report() -> tuple[CompareReport, Path, Path]: + project_root = Path(__file__).resolve().parents[1] + examples = project_root / "examples" + before_path = examples / "requirements_before.txt" + after_path = examples / "requirements_after.txt" + + urllib3 = Component( + name="urllib3", + version="2.2.1", + ecosystem="pypi", + purl="pkg:pypi/urllib3@2.2.1", + source_url="https://pypi.org/project/urllib3/2.2.1/", + provenance=_provenance( + name="urllib3", + version="2.2.1", + statuses=(ProvenanceStatus.PROVENANCE_AVAILABLE, ProvenanceStatus.ATTESTATION_AVAILABLE), + file_statuses=(ProvenanceStatus.PROVENANCE_AVAILABLE, ProvenanceStatus.ATTESTATION_AVAILABLE), + attestation_count=1, + publisher_kinds=("github actions",), + predicate_types=("https://example.test/attestation/v1",), + ), + ) + mystery_lib = Component( + name="mystery-lib", + version="1.0.0", + ecosystem="pypi", + purl="pkg:pypi/mystery-lib@1.0.0", + source_url="https://pypi.org/project/mystery-lib/1.0.0/", + provenance=_provenance( + name="mystery-lib", + version="1.0.0", + statuses=(ProvenanceStatus.ATTESTATION_UNAVAILABLE,), + file_statuses=(ProvenanceStatus.ATTESTATION_UNAVAILABLE,), + attestation_count=0, + ), + ) + legacy_before = Component( + name="legacy-lib", + version="1.0.0", + ecosystem="pypi", + purl="pkg:pypi/legacy-lib@1.0.0", + ) + legacy_after = Component( + name="legacy-lib", + version="1.1.0", + ecosystem="pypi", + purl="pkg:pypi/legacy-lib@1.1.0", + provenance=_provenance( + name="legacy-lib", + version="1.1.0", + statuses=(ProvenanceStatus.PROVENANCE_AVAILABLE, ProvenanceStatus.ATTESTATION_AVAILABLE), + file_statuses=(ProvenanceStatus.PROVENANCE_AVAILABLE, ProvenanceStatus.ATTESTATION_AVAILABLE), + attestation_count=1, + publisher_kinds=("manual upload",), + predicate_types=("https://example.test/attestation/v1",), + ), + ) + legacy_change = ComponentChange( + key=component_key(legacy_after), + before=legacy_before, + after=legacy_after, + classification="version_changed", + ) + + risks = [ + RiskFinding( + bucket=RiskBucket.NEW_PACKAGE, + component_key=component_key(urllib3), + component=urllib3, + rationale="Component was not present in the before input.", + ), + RiskFinding( + bucket=RiskBucket.NEW_PACKAGE, + component_key=component_key(mystery_lib), + component=mystery_lib, + rationale="Component was not present in the before input.", + ), + RiskFinding( + bucket=RiskBucket.VERSION_CHANGE_UNCLASSIFIED, + component_key=component_key(legacy_after), + component=legacy_after, + rationale="Version changed but did not qualify as a parseable SemVer major upgrade.", + ), + ] + policy = PolicyConfig( + version=2, + block_on=("provenance_required", "unverified_provenance"), + warn_on=("missing_attestation",), + require_attestations_for_new_packages=True, + allow_provenance_publishers=("github actions",), + ) + policy_evaluation = evaluate_policy( + policy, + policy_path="examples/policy-provenance-strict.yml", + added=[urllib3, mystery_lib], + changed=[legacy_change], + findings=risks, + ) + report = CompareReport( + summary=ReportSummary( + added=2, + removed=0, + changed=1, + risk_counts=summarize_risks(risks), + ), + components=ReportComponents( + added=[urllib3, mystery_lib], + removed=[], + changed=[legacy_change], + ), + risks=risks, + metadata=ReportMetadata( + before_format="requirements-txt", + after_format="requirements-txt", + generated_at=None, + strict=False, + stub=False, + policy_evaluation=policy_evaluation, + enrichment=ReportEnrichmentMetadata( + mode="opt_in_pypi", + pypi_enabled=True, + pypi_timeout_seconds=3.0, + pypi_network_access_performed=True, + network_access_performed=True, + candidate_components=3, + supported_components=3, + status_counts={ + "attestation_available": 2, + "attestation_unavailable": 1, + "provenance_available": 2, + }, + ), + ), + notes=[ + "This tool uses heuristic risk classification.", + "PyPI provenance enrichment was requested explicitly.", + ], + ) + return report, before_path, after_path + + +def _provenance( + *, + name: str, + version: str, + statuses: tuple[ProvenanceStatus, ...], + file_statuses: tuple[ProvenanceStatus, ...], + attestation_count: int, + publisher_kinds: tuple[str, ...] = (), + predicate_types: tuple[str, ...] = (), +) -> ProvenanceEvidence: + return ProvenanceEvidence( + provider="pypi", + requested=True, + supported=True, + lookup_performed=True, + package_name=name, + package_version=version, + release_url=f"https://pypi.org/project/{name}/{version}/", + statuses=statuses, + files=( + ProvenanceFileEvidence( + filename=f"{name}-{version}.tar.gz", + url=f"https://files.pythonhosted.org/packages/{name}-{version}.tar.gz", + sha256="deadbeef", + upload_time="2026-04-01T00:00:00.000000Z", + yanked=False, + statuses=file_statuses, + attestation_count=attestation_count, + predicate_types=predicate_types, + publisher_kinds=publisher_kinds, + ), + ), + files_evaluated=1, + files_with_attestations=1 if attestation_count > 0 else 0, + files_without_attestations=0 if attestation_count > 0 else 1, + ) + + +def _read_example(name: str) -> str: + examples = Path(__file__).resolve().parents[1] / "examples" + return (examples / name).read_text(encoding="utf-8") + + +def _normalize_sarif_snapshot(content: str) -> str: + payload = json.loads(content) + payload["runs"][0]["originalUriBaseIds"]["%SRCROOT%"]["uri"] = "file:///__PROJECT_ROOT__/" + return json.dumps(payload, indent=2) + "\n" diff --git a/tools/sbom-diff-and-risk/tests/test_pypi_enrichment.py b/tools/sbom-diff-and-risk/tests/test_pypi_enrichment.py new file mode 100644 index 0000000..73617fb --- /dev/null +++ b/tools/sbom-diff-and-risk/tests/test_pypi_enrichment.py @@ -0,0 +1,265 @@ +from __future__ import annotations + +import base64 +import json + +from sbom_diff_risk.enrichment import PyPIProvenanceEnricher, normalize_provenance_file +from sbom_diff_risk.models import Component, ProvenanceEvidence, ProvenanceStatus +from sbom_diff_risk.pypi_integrity_client import ( + PyPIAttestation, + PyPIClientError, + PyPIFileProvenance, + PyPIRelease, + PyPIReleaseFile, +) +from sbom_diff_risk.pypi_provenance import provenance_evidence_to_dict + + +def test_normalize_provenance_file_decodes_predicate_type() -> None: + release_file = _release_file() + provenance = PyPIFileProvenance( + filename=release_file.filename, + attestation_count=1, + attestations=( + PyPIAttestation( + statement=_encoded_statement({"predicateType": "https://example.test/attestation/v1"}), + publisher_kind="GitHub Actions", + ), + ), + ) + + normalized = normalize_provenance_file(release_file=release_file, provenance=provenance) + + assert normalized.statuses == ( + ProvenanceStatus.PROVENANCE_AVAILABLE, + ProvenanceStatus.ATTESTATION_AVAILABLE, + ) + assert normalized.predicate_types == ("https://example.test/attestation/v1",) + assert normalized.publisher_kinds == ("GitHub Actions",) + + +def test_enricher_records_attestation_available_for_supported_package() -> None: + client = FakePyPIClient( + releases={ + ("requests", "2.31.0"): PyPIRelease( + project="requests", + version="2.31.0", + release_url="https://pypi.org/project/requests/2.31.0/", + files=(_release_file(filename="requests-2.31.0.tar.gz"),), + ) + }, + provenance={ + ("requests", "2.31.0", "requests-2.31.0.tar.gz"): PyPIFileProvenance( + filename="requests-2.31.0.tar.gz", + attestation_count=1, + attestations=( + PyPIAttestation( + statement=_encoded_statement({"predicateType": "https://example.test/attestation/v1"}), + publisher_kind="GitHub Actions", + ), + ), + ) + }, + ) + enricher = PyPIProvenanceEnricher(client=client, timeout_seconds=2.5) + + [component] = enricher.enrich_components([Component(name="requests", version="2.31.0", ecosystem="pypi")]) + metadata = enricher.build_report_metadata() + + assert component.provenance is not None + assert component.provenance.statuses == ( + ProvenanceStatus.PROVENANCE_AVAILABLE, + ProvenanceStatus.ATTESTATION_AVAILABLE, + ) + assert component.provenance.supported is True + assert component.provenance.lookup_performed is True + assert component.provenance.files_evaluated == 1 + assert component.provenance.files_with_attestations == 1 + assert component.provenance.files_without_attestations == 0 + assert component.provenance.files[0].attestation_count == 1 + assert metadata.mode == "opt_in_pypi" + assert metadata.network_access_performed is True + assert metadata.supported_components == 1 + assert metadata.status_counts == { + "attestation_available": 1, + "provenance_available": 1, + } + + +def test_enricher_marks_attestation_unavailable_when_provenance_endpoint_returns_none() -> None: + client = FakePyPIClient( + releases={ + ("urllib3", "2.2.1"): PyPIRelease( + project="urllib3", + version="2.2.1", + release_url="https://pypi.org/project/urllib3/2.2.1/", + files=(_release_file(filename="urllib3-2.2.1.tar.gz"),), + ) + }, + provenance={ + ("urllib3", "2.2.1", "urllib3-2.2.1.tar.gz"): None, + }, + ) + enricher = PyPIProvenanceEnricher(client=client) + + [component] = enricher.enrich_components([Component(name="urllib3", version="2.2.1", ecosystem="pypi")]) + + assert component.provenance is not None + assert component.provenance.statuses == (ProvenanceStatus.ATTESTATION_UNAVAILABLE,) + assert component.provenance.supported is True + assert component.provenance.lookup_performed is True + assert component.provenance.files_evaluated == 1 + assert component.provenance.files_with_attestations == 0 + assert component.provenance.files_without_attestations == 1 + assert component.provenance.files[0].statuses == (ProvenanceStatus.ATTESTATION_UNAVAILABLE,) + + +def test_enricher_marks_unsupported_for_non_pypi_component_without_network_access() -> None: + client = FakePyPIClient() + enricher = PyPIProvenanceEnricher(client=client) + + [component] = enricher.enrich_components([Component(name="left-pad", version="1.3.0", ecosystem="npm")]) + metadata = enricher.build_report_metadata() + + assert component.provenance is not None + assert component.provenance.statuses == (ProvenanceStatus.UNSUPPORTED_FOR_PACKAGE,) + assert component.provenance.supported is False + assert component.provenance.lookup_performed is False + assert client.release_calls == [] + assert metadata.network_access_performed is False + assert metadata.supported_components == 0 + assert metadata.status_counts == {"unsupported_for_package": 1} + + +def test_enricher_captures_timeout_error_as_evidence() -> None: + client = FakePyPIClient( + release_errors={ + ("certifi", "2026.1.1"): PyPIClientError( + "PyPI request timed out after 2.5 seconds for https://pypi.org/pypi/certifi/2026.1.1/json.", + is_timeout=True, + ) + } + ) + enricher = PyPIProvenanceEnricher(client=client, timeout_seconds=2.5) + + [component] = enricher.enrich_components([Component(name="certifi", version="2026.1.1", ecosystem="pypi")]) + metadata = enricher.build_report_metadata() + + assert component.provenance is not None + assert component.provenance.statuses == (ProvenanceStatus.ENRICHMENT_ERROR,) + assert "timed out" in (component.provenance.error or "") + assert metadata.network_access_performed is True + assert metadata.status_counts == {"enrichment_error": 1} + + +def test_enricher_records_lookup_when_release_is_missing_after_explicit_opt_in() -> None: + client = FakePyPIClient( + release_errors={ + ("ghost-package", "9.9.9"): PyPIClientError( + "PyPI request failed with HTTP 404 for https://pypi.org/pypi/ghost-package/9.9.9/json.", + status_code=404, + ) + } + ) + enricher = PyPIProvenanceEnricher(client=client) + + [component] = enricher.enrich_components([Component(name="ghost-package", version="9.9.9", ecosystem="pypi")]) + metadata = enricher.build_report_metadata() + + assert component.provenance is not None + assert component.provenance.statuses == (ProvenanceStatus.UNSUPPORTED_FOR_PACKAGE,) + assert component.provenance.supported is False + assert component.provenance.lookup_performed is True + assert client.release_calls == [("ghost-package", "9.9.9")] + assert metadata.network_access_performed is True + assert metadata.supported_components == 0 + assert metadata.status_counts == {"unsupported_for_package": 1} + + +def test_provenance_evidence_to_dict_includes_lookup_and_file_counts() -> None: + provenance = ProvenanceEvidence( + provider="pypi", + requested=True, + supported=True, + lookup_performed=True, + package_name="requests", + package_version="2.31.0", + release_url="https://pypi.org/project/requests/2.31.0/", + statuses=( + ProvenanceStatus.PROVENANCE_AVAILABLE, + ProvenanceStatus.ATTESTATION_AVAILABLE, + ), + files=( + normalize_provenance_file( + release_file=_release_file(filename="requests-2.31.0.tar.gz"), + provenance=PyPIFileProvenance( + filename="requests-2.31.0.tar.gz", + attestation_count=1, + attestations=( + PyPIAttestation( + statement=_encoded_statement({"predicateType": "https://example.test/attestation/v1"}), + publisher_kind="GitHub Actions", + ), + ), + ), + ), + ), + files_evaluated=1, + files_with_attestations=1, + files_without_attestations=0, + ) + + payload = provenance_evidence_to_dict(provenance) + + assert payload is not None + assert payload["supported"] is True + assert payload["lookup_performed"] is True + assert payload["files_evaluated"] == 1 + assert payload["files_with_attestations"] == 1 + assert payload["files_without_attestations"] == 0 + + +class FakePyPIClient: + def __init__( + self, + *, + releases: dict[tuple[str, str], PyPIRelease] | None = None, + provenance: dict[tuple[str, str, str], PyPIFileProvenance | None] | None = None, + release_errors: dict[tuple[str, str], Exception] | None = None, + provenance_errors: dict[tuple[str, str, str], Exception] | None = None, + ) -> None: + self._releases = releases or {} + self._provenance = provenance or {} + self._release_errors = release_errors or {} + self._provenance_errors = provenance_errors or {} + self.release_calls: list[tuple[str, str]] = [] + self.provenance_calls: list[tuple[str, str, str]] = [] + + def fetch_release(self, project: str, version: str) -> PyPIRelease: + key = (project, version) + self.release_calls.append(key) + if key in self._release_errors: + raise self._release_errors[key] + return self._releases[key] + + def fetch_provenance(self, project: str, version: str, filename: str) -> PyPIFileProvenance | None: + key = (project, version, filename) + self.provenance_calls.append(key) + if key in self._provenance_errors: + raise self._provenance_errors[key] + return self._provenance[key] + + +def _release_file(*, filename: str = "example-1.0.0.tar.gz") -> PyPIReleaseFile: + return PyPIReleaseFile( + filename=filename, + url=f"https://files.pythonhosted.org/packages/source/{filename}", + sha256="deadbeef", + upload_time="2026-04-01T00:00:00.000000Z", + yanked=False, + ) + + +def _encoded_statement(payload: dict[str, object]) -> str: + encoded = base64.urlsafe_b64encode(json.dumps(payload).encode("utf-8")).decode("utf-8") + return encoded.rstrip("=") diff --git a/tools/sbom-diff-and-risk/tests/test_pypi_integrity_client.py b/tools/sbom-diff-and-risk/tests/test_pypi_integrity_client.py new file mode 100644 index 0000000..495a453 --- /dev/null +++ b/tools/sbom-diff-and-risk/tests/test_pypi_integrity_client.py @@ -0,0 +1,98 @@ +from __future__ import annotations + +import io +import json +from urllib import error + +from sbom_diff_risk.pypi_integrity_client import ( + PyPIIntegrityClient, + parse_provenance_payload, + parse_release_payload, +) + + +def test_parse_release_payload_extracts_sorted_files() -> None: + payload = { + "info": { + "package_url": "https://pypi.org/project/example/", + }, + "urls": [ + { + "filename": "example-1.0.0.tar.gz", + "url": "https://files.pythonhosted.org/packages/source/e/example/example-1.0.0.tar.gz", + "digests": {"sha256": "bbb"}, + "upload_time_iso_8601": "2026-04-01T00:00:00.000000Z", + "yanked": False, + }, + { + "filename": "example-1.0.0-py3-none-any.whl", + "url": "https://files.pythonhosted.org/packages/example-1.0.0-py3-none-any.whl", + "digests": {"sha256": "aaa"}, + "upload_time_iso_8601": "2026-04-01T00:00:01.000000Z", + "yanked": False, + }, + ], + } + + release = parse_release_payload(payload, project="example", version="1.0.0") + + assert release.release_url == "https://pypi.org/project/example/" + assert [item.filename for item in release.files] == [ + "example-1.0.0-py3-none-any.whl", + "example-1.0.0.tar.gz", + ] + assert release.files[0].sha256 == "aaa" + + +def test_parse_provenance_payload_flattens_attestations() -> None: + payload = { + "attestation_bundles": [ + { + "publisher": {"kind": "GitHub Actions"}, + "attestations": [ + {"envelope": {"statement": "eyJwcmVkaWNhdGVUeXBlIjoiaHR0cHM6Ly9leGFtcGxlLmNvbS9wcmVkaWNhdGUifQ"}} + ], + } + ] + } + + provenance = parse_provenance_payload(payload, filename="example-1.0.0.tar.gz") + + assert provenance.filename == "example-1.0.0.tar.gz" + assert provenance.attestation_count == 1 + assert provenance.attestations[0].publisher_kind == "GitHub Actions" + + +def test_fetch_provenance_returns_none_for_404() -> None: + url = "https://pypi.org/integrity/example/1.0.0/example-1.0.0.tar.gz/provenance" + client = PyPIIntegrityClient(opener=FakeOpener({url: _http_error(url, 404)})) + + provenance = client.fetch_provenance("example", "1.0.0", "example-1.0.0.tar.gz") + + assert provenance is None + + +class FakeOpener: + def __init__(self, responses: dict[str, object]) -> None: + self._responses = responses + + def open(self, req, timeout: float): # noqa: ANN001 + outcome = self._responses[req.full_url] + if isinstance(outcome, Exception): + raise outcome + return _Response(outcome) + + +class _Response(io.BytesIO): + def __init__(self, payload: object) -> None: + super().__init__(json.dumps(payload).encode("utf-8")) + + def __enter__(self): # noqa: ANN204 + return self + + def __exit__(self, exc_type, exc, tb) -> None: # noqa: ANN001 + self.close() + + +def _http_error(url: str, code: int) -> error.HTTPError: + return error.HTTPError(url, code, "error", hdrs=None, fp=io.BytesIO(b"{}")) diff --git a/tools/sbom-diff-and-risk/tests/test_reports.py b/tools/sbom-diff-and-risk/tests/test_reports.py index d718cf4..8f3c4b2 100644 --- a/tools/sbom-diff-and-risk/tests/test_reports.py +++ b/tools/sbom-diff-and-risk/tests/test_reports.py @@ -4,7 +4,16 @@ from pathlib import Path from sbom_diff_risk.diffing import diff_components -from sbom_diff_risk.models import CompareReport, ReportComponents, ReportMetadata, ReportSummary +from sbom_diff_risk.models import ( + CompareReport, + Component, + ProvenanceEvidence, + ProvenanceFileEvidence, + ProvenanceStatus, + ReportComponents, + ReportMetadata, + ReportSummary, +) from sbom_diff_risk.policy_evaluator import evaluate_policy from sbom_diff_risk.policy_models import PolicyConfig from sbom_diff_risk.policy_parser import build_policy @@ -100,10 +109,54 @@ def test_report_json_keeps_legacy_sections() -> None: "warning_findings", "suppressed_findings", "rule_catalog", + "provenance_summary", + "attestation_summary", + "scorecard_summary", + "enrichment_metadata", + "trust_signal_notes", "metadata", "notes", } assert payload["metadata"]["policy_evaluation"] == payload["policy_evaluation"] + assert payload["metadata"]["enrichment"] == payload["enrichment_metadata"] + assert "provenance_policy" not in payload + assert "provenance_policy_impact" not in payload + + +def test_report_json_offline_enrichment_metadata_is_stable_by_default() -> None: + report = _build_report("cdx_before.json", "cdx_after.json") + + first = render_report_json(report) + second = render_report_json(report) + payload = json.loads(first) + + assert first == second + assert payload["metadata"]["enrichment"] == { + "mode": "offline_default", + "pypi_enabled": False, + "pypi_timeout_seconds": None, + "pypi_network_access_performed": False, + "network_access_performed": False, + "candidate_components": 0, + "supported_components": 0, + "status_counts": {}, + "scorecard_enabled": False, + "scorecard_timeout_seconds": None, + "scorecard_network_access_performed": False, + "scorecard_candidate_components": 0, + "scorecard_supported_components": 0, + "scorecard_status_counts": {}, + } + assert payload["provenance_summary"]["pypi_components_without_provenance"] == 2 + assert payload["provenance_summary"]["components_with_provenance"] == 0 + assert payload["attestation_summary"]["files_evaluated"] == 0 + assert payload["scorecard_summary"]["enabled"] is False + assert payload["scorecard_summary"]["components_with_scorecards"] == 0 + assert payload["scorecard_summary"]["repository_unmapped"] == 0 + assert payload["trust_signal_notes"] == ["PyPI components are present, but provenance enrichment was not enabled for this run."] + added_components = payload["components"]["added"] + assert all("provenance" not in component["evidence"] for component in added_components) + assert all("scorecard" not in component["evidence"] for component in added_components) def test_reports_render_suppressions_when_policy_ignores_findings() -> None: @@ -122,6 +175,64 @@ def test_reports_render_suppressions_when_policy_ignores_findings() -> None: assert "## Suppressions" in markdown +def test_reports_include_provenance_policy_details_for_v2_policy() -> None: + component = Component( + name="urllib3", + version="2.2.1", + ecosystem="pypi", + provenance=ProvenanceEvidence( + provider="pypi", + requested=True, + package_name="urllib3", + package_version="2.2.1", + release_url="https://pypi.org/project/urllib3/2.2.1/", + statuses=(ProvenanceStatus.ATTESTATION_UNAVAILABLE,), + files=( + ProvenanceFileEvidence( + filename="urllib3-2.2.1.tar.gz", + statuses=(ProvenanceStatus.ATTESTATION_UNAVAILABLE,), + attestation_count=0, + ), + ), + ), + ) + policy = PolicyConfig( + version=2, + warn_on=("missing_attestation",), + require_attestations_for_new_packages=True, + allow_unattested_packages=("pip",), + ) + policy_evaluation = evaluate_policy(policy, policy_path="policy-provenance-minimal.yml", added=[component], changed=[], findings=[]) + report = CompareReport( + summary=ReportSummary(added=1, removed=0, changed=0, risk_counts={}), + components=ReportComponents(added=[component], removed=[], changed=[]), + risks=[], + metadata=ReportMetadata( + before_format="requirements-txt", + after_format="requirements-txt", + generated_at=None, + strict=False, + stub=False, + policy_evaluation=policy_evaluation, + ), + notes=["PyPI provenance enrichment was requested explicitly."], + ) + + payload = json.loads(render_report_json(report)) + markdown = render_report_markdown(report) + + assert payload["policy_evaluation"]["effective_policy"]["require_attestations_for_new_packages"] is True + assert payload["policy_evaluation"]["effective_policy"]["allow_unattested_packages"] == ["pip"] + assert payload["provenance_policy"]["configured"] is True + assert payload["provenance_policy"]["requirements"]["allow_unattested_packages"] == ["pip"] + assert payload["provenance_policy"]["counts"]["blocking"] == 1 + assert any(item["rule_id"] == "provenance_required" for item in payload["blocking_findings"]) + assert "- Configured provenance policy: yes" in markdown + assert "- Allow unattested packages: pip" in markdown + assert "provenance_required" in markdown + assert "missing_attestation" in markdown + + def _build_report( before_name: str, after_name: str, @@ -143,7 +254,7 @@ def _build_report( if policy is None: built_policy, policy_path = build_policy( - policy_path=(Path("examples") / policy_name) if policy_name else None, + policy_path=examples / policy_name if policy_name else None, fail_on=fail_on, warn_on=warn_on, ) diff --git a/tools/sbom-diff-and-risk/tests/test_repository_mapping.py b/tools/sbom-diff-and-risk/tests/test_repository_mapping.py new file mode 100644 index 0000000..a1994df --- /dev/null +++ b/tools/sbom-diff-and-risk/tests/test_repository_mapping.py @@ -0,0 +1,148 @@ +from __future__ import annotations + +from sbom_diff_risk.models import Component +from sbom_diff_risk.repository_mapping import map_component_to_repository + + +def test_map_component_to_repository_accepts_direct_github_repository_url() -> None: + component = Component( + name="requests", + version="2.32.0", + ecosystem="pypi", + source_url="https://github.com/psf/requests", + evidence={"source_format": "cyclonedx-json"}, + ) + + mapping = map_component_to_repository(component) + + assert mapping is not None + assert mapping.canonical_name == "github.com/psf/requests" + assert mapping.source == "component.source_url" + + +def test_map_component_to_repository_prefers_explicit_vcs_reference() -> None: + component = Component( + name="requests", + version="2.32.0", + ecosystem="pypi", + source_url="https://pypi.org/project/requests/", + evidence={ + "source_format": "cyclonedx-json", + "component": { + "externalReferences": [ + {"type": "website", "url": "https://pypi.org/project/requests/"}, + {"type": "vcs", "url": "https://github.com/psf/requests"}, + ] + }, + }, + ) + + mapping = map_component_to_repository(component) + + assert mapping is not None + assert mapping.canonical_name == "github.com/psf/requests" + assert mapping.source == "cyclonedx.externalReferences.vcs" + + +def test_map_component_to_repository_rejects_website_repository_hint_as_low_confidence() -> None: + component = Component( + name="requests", + version="2.32.0", + ecosystem="pypi", + source_url="https://github.com/psf/requests", + evidence={ + "source_format": "cyclonedx-json", + "component": { + "externalReferences": [ + {"type": "website", "url": "https://github.com/psf/requests"}, + ] + }, + }, + ) + + assert map_component_to_repository(component) is None + + +def test_map_component_to_repository_rejects_registry_url_without_explicit_repo() -> None: + component = Component( + name="urllib3", + version="2.2.1", + ecosystem="pypi", + source_url="https://pypi.org/project/urllib3/2.2.1/", + evidence={"source_format": "requirements-txt"}, + ) + + assert map_component_to_repository(component) is None + + +def test_map_component_to_repository_rejects_deep_repository_paths() -> None: + component = Component( + name="requests", + version="2.32.0", + ecosystem="pypi", + source_url="https://github.com/psf/requests/tree/main", + evidence={"source_format": "cyclonedx-json"}, + ) + + assert map_component_to_repository(component) is None + + +def test_map_component_to_repository_rejects_ambiguous_explicit_repositories() -> None: + component = Component( + name="example", + version="1.0.0", + ecosystem="pypi", + evidence={ + "source_format": "cyclonedx-json", + "component": { + "externalReferences": [ + {"type": "vcs", "url": "https://github.com/example/one"}, + {"type": "vcs", "url": "https://github.com/example/two"}, + ] + }, + }, + ) + + assert map_component_to_repository(component) is None + + +def test_map_component_to_repository_accepts_explicit_spdx_vcs_reference() -> None: + component = Component( + name="requests", + version="2.32.0", + ecosystem="pypi", + evidence={ + "source_format": "spdx-json", + "package": { + "externalRefs": [ + { + "referenceType": "vcs", + "referenceLocator": "https://github.com/psf/requests", + } + ] + }, + }, + ) + + mapping = map_component_to_repository(component) + + assert mapping is not None + assert mapping.canonical_name == "github.com/psf/requests" + assert mapping.source == "spdx.externalRefs.vcs" + + +def test_map_component_to_repository_rejects_spdx_homepage_repository_hint_as_low_confidence() -> None: + component = Component( + name="requests", + version="2.32.0", + ecosystem="pypi", + source_url="https://github.com/psf/requests", + evidence={ + "source_format": "spdx-json", + "package": { + "homepage": "https://github.com/psf/requests", + }, + }, + ) + + assert map_component_to_repository(component) is None diff --git a/tools/sbom-diff-and-risk/tests/test_sarif.py b/tools/sbom-diff-and-risk/tests/test_sarif.py index 5eb5c7b..624d173 100644 --- a/tools/sbom-diff-and-risk/tests/test_sarif.py +++ b/tools/sbom-diff-and-risk/tests/test_sarif.py @@ -2,7 +2,6 @@ import argparse import json -import re from pathlib import Path from sbom_diff_risk.cli import run_compare @@ -31,7 +30,7 @@ def test_render_report_sarif_matches_golden() -> None: rendered = render_report_sarif(report, before_path=before_path, after_path=after_path, base_dir=project_root) expected = (project_root / "examples" / "sample-sarif.sarif").read_text(encoding="utf-8") - assert _normalize_sarif_golden(rendered) == _normalize_sarif_golden(expected) + assert _normalize_sarif_snapshot(rendered) == _normalize_sarif_snapshot(expected) def test_sarif_rule_ids_are_stable() -> None: @@ -172,7 +171,7 @@ def _build_report(before_name: str, after_name: str, *, policy_name: str | None added, removed, changed = diff_components(before_components, after_components) risks = evaluate_risks(added, changed, allowlist=["pypi.org", "files.pythonhosted.org", "github.com"]) - policy, policy_path = build_policy(policy_path=(Path("examples") / policy_name) if policy_name else None) + policy, policy_path = build_policy(policy_path=examples / policy_name if policy_name else None) policy_evaluation = evaluate_policy( policy, policy_path=policy_path, @@ -213,9 +212,7 @@ def _build_report(before_name: str, after_name: str, *, policy_name: str | None return report, before_path, after_path -def _normalize_sarif_golden(value: str) -> str: - return re.sub( - r"file:///[^\"\r\n]+/tools/sbom-diff-and-risk(?:-real)?/", - "file:///__PROJECT_ROOT__/", - value, - ) +def _normalize_sarif_snapshot(content: str) -> str: + payload = json.loads(content) + payload["runs"][0]["originalUriBaseIds"]["%SRCROOT%"]["uri"] = "file:///__PROJECT_ROOT__/" + return json.dumps(payload, indent=2) + "\n" diff --git a/tools/sbom-diff-and-risk/tests/test_scorecard_client.py b/tools/sbom-diff-and-risk/tests/test_scorecard_client.py new file mode 100644 index 0000000..0e47a6a --- /dev/null +++ b/tools/sbom-diff-and-risk/tests/test_scorecard_client.py @@ -0,0 +1,86 @@ +from __future__ import annotations + +import io +import json +from urllib import error + +import pytest + +from sbom_diff_risk.scorecard_client import ( + ScorecardClient, + ScorecardClientError, + parse_project_payload, +) + + +def test_parse_project_payload_extracts_score_and_checks() -> None: + payload = { + "date": "2026-04-10T00:00:00Z", + "repo": { + "name": "github.com/psf/requests", + "commit": "abc123", + }, + "scorecard": { + "version": "5.0.0", + "commit": "def456", + }, + "score": 7.8, + "checks": [ + { + "name": "Maintained", + "score": 10, + "reason": "Project is active.", + "documentation": {"url": "https://example.test/docs/maintained", "short": "Maintained"}, + }, + { + "name": "Branch-Protection", + "score": 6, + "reason": "Not all protections are enabled.", + }, + ], + } + + result = parse_project_payload(payload, expected_canonical_name="github.com/psf/requests") + + assert result.canonical_name == "github.com/psf/requests" + assert result.score == 7.8 + assert result.repository_commit == "abc123" + assert result.scorecard_version == "5.0.0" + assert [check.name for check in result.checks] == ["Branch-Protection", "Maintained"] + assert result.checks[1].documentation_url == "https://example.test/docs/maintained" + + +def test_scorecard_client_surfaces_404_as_client_error() -> None: + url = "https://api.securityscorecards.dev/projects/github.com/psf/requests" + client = ScorecardClient(opener=FakeOpener({url: _http_error(url, 404)})) + + with pytest.raises(ScorecardClientError) as excinfo: + client.fetch_project("github.com", "psf", "requests") + + assert excinfo.value.status_code == 404 + + +class FakeOpener: + def __init__(self, responses: dict[str, object]) -> None: + self._responses = responses + + def open(self, req, timeout: float): # noqa: ANN001 + outcome = self._responses[req.full_url] + if isinstance(outcome, Exception): + raise outcome + return _Response(outcome) + + +class _Response(io.BytesIO): + def __init__(self, payload: object) -> None: + super().__init__(json.dumps(payload).encode("utf-8")) + + def __enter__(self): # noqa: ANN204 + return self + + def __exit__(self, exc_type, exc, tb) -> None: # noqa: ANN001 + self.close() + + +def _http_error(url: str, code: int) -> error.HTTPError: + return error.HTTPError(url, code, "error", hdrs=None, fp=io.BytesIO(b"{}")) diff --git a/tools/sbom-diff-and-risk/tests/test_scorecard_enrichment.py b/tools/sbom-diff-and-risk/tests/test_scorecard_enrichment.py new file mode 100644 index 0000000..3ff640f --- /dev/null +++ b/tools/sbom-diff-and-risk/tests/test_scorecard_enrichment.py @@ -0,0 +1,197 @@ +from __future__ import annotations + +from sbom_diff_risk.models import Component, ScorecardCheck, ScorecardStatus +from sbom_diff_risk.scorecard_client import ScorecardClientError, ScorecardProjectResult +from sbom_diff_risk.scorecard_enrichment import ScorecardEnricher + + +def test_scorecard_enricher_records_available_scorecard_for_mapped_repository() -> None: + client = FakeScorecardClient( + responses={ + ("github.com", "psf", "requests"): ScorecardProjectResult( + canonical_name="github.com/psf/requests", + score=7.8, + date="2026-04-10T00:00:00Z", + scorecard_version="5.0.0", + scorecard_commit="def456", + repository_commit="abc123", + checks=( + ScorecardCheck(name="Maintained", score=10, reason="Project is active."), + ScorecardCheck(name="Branch-Protection", score=6, reason="Not all protections are enabled."), + ), + ) + } + ) + enricher = ScorecardEnricher(client=client, timeout_seconds=2.5) + + [component] = enricher.enrich_components( + [Component(name="requests", version="2.32.0", ecosystem="pypi", source_url="https://github.com/psf/requests")] + ) + metadata = enricher.build_report_metadata() + + assert component.scorecard is not None + assert component.scorecard.statuses == (ScorecardStatus.SCORECARD_AVAILABLE,) + assert component.scorecard.repository is not None + assert component.scorecard.repository.canonical_name == "github.com/psf/requests" + assert component.scorecard.score == 7.8 + assert client.calls == [("github.com", "psf", "requests")] + assert metadata.mode == "opt_in_scorecard" + assert metadata.scorecard_enabled is True + assert metadata.scorecard_network_access_performed is True + assert metadata.scorecard_status_counts == {"scorecard_available": 1} + + +def test_scorecard_enricher_marks_repository_unmapped_without_network_access() -> None: + client = FakeScorecardClient() + enricher = ScorecardEnricher(client=client) + + [component] = enricher.enrich_components( + [Component(name="urllib3", version="2.2.1", ecosystem="pypi", source_url="https://pypi.org/project/urllib3/2.2.1/")] + ) + metadata = enricher.build_report_metadata() + + assert component.scorecard is not None + assert component.scorecard.statuses == (ScorecardStatus.REPOSITORY_UNMAPPED,) + assert client.calls == [] + assert metadata.scorecard_network_access_performed is False + assert metadata.scorecard_supported_components == 0 + assert metadata.scorecard_status_counts == {"repository_unmapped": 1} + + +def test_scorecard_enricher_skips_low_confidence_repository_hints_without_network_access() -> None: + client = FakeScorecardClient() + enricher = ScorecardEnricher(client=client) + + [component] = enricher.enrich_components( + [ + Component( + name="requests", + version="2.32.0", + ecosystem="pypi", + source_url="https://github.com/psf/requests", + evidence={ + "source_format": "cyclonedx-json", + "component": { + "externalReferences": [ + {"type": "website", "url": "https://github.com/psf/requests"}, + ] + }, + }, + ) + ] + ) + metadata = enricher.build_report_metadata() + + assert component.scorecard is not None + assert component.scorecard.statuses == (ScorecardStatus.REPOSITORY_UNMAPPED,) + assert client.calls == [] + assert metadata.scorecard_network_access_performed is False + assert metadata.scorecard_status_counts == {"repository_unmapped": 1} + + +def test_scorecard_enricher_marks_scorecard_unavailable_for_404() -> None: + client = FakeScorecardClient( + errors={ + ("github.com", "psf", "requests"): ScorecardClientError( + "Scorecard request failed with HTTP 404 for https://api.securityscorecards.dev/projects/github.com/psf/requests.", + status_code=404, + ) + } + ) + enricher = ScorecardEnricher(client=client) + + [component] = enricher.enrich_components( + [Component(name="requests", version="2.32.0", ecosystem="pypi", source_url="https://github.com/psf/requests")] + ) + + assert component.scorecard is not None + assert component.scorecard.statuses == (ScorecardStatus.SCORECARD_UNAVAILABLE,) + assert component.scorecard.note == "Scorecard data is not available for the mapped repository." + + +def test_scorecard_enricher_captures_timeout_as_enrichment_error() -> None: + client = FakeScorecardClient( + errors={ + ("github.com", "psf", "requests"): ScorecardClientError( + "Scorecard request timed out after 2.5 seconds for https://api.securityscorecards.dev/projects/github.com/psf/requests.", + is_timeout=True, + ) + } + ) + enricher = ScorecardEnricher(client=client, timeout_seconds=2.5) + + [component] = enricher.enrich_components( + [Component(name="requests", version="2.32.0", ecosystem="pypi", source_url="https://github.com/psf/requests")] + ) + metadata = enricher.build_report_metadata() + + assert component.scorecard is not None + assert component.scorecard.statuses == (ScorecardStatus.ENRICHMENT_ERROR,) + assert "timed out" in (component.scorecard.error or "") + assert metadata.scorecard_network_access_performed is True + assert metadata.scorecard_status_counts == {"enrichment_error": 1} + + +def test_scorecard_enricher_re_evaluates_same_package_identity_when_repository_hints_differ() -> None: + client = FakeScorecardClient( + responses={ + ("github.com", "psf", "requests"): ScorecardProjectResult( + canonical_name="github.com/psf/requests", + score=7.8, + date="2026-04-10T00:00:00Z", + scorecard_version="5.0.0", + scorecard_commit="def456", + repository_commit="abc123", + checks=(ScorecardCheck(name="Maintained", score=10, reason="Project is active."),), + ) + } + ) + enricher = ScorecardEnricher(client=client) + + weak_component = Component( + name="requests", + version="2.32.0", + ecosystem="pypi", + source_url="https://github.com/psf/requests", + evidence={ + "source_format": "cyclonedx-json", + "component": { + "externalReferences": [ + {"type": "website", "url": "https://github.com/psf/requests"}, + ] + }, + }, + ) + strong_component = Component( + name="requests", + version="2.32.0", + ecosystem="pypi", + source_url="https://github.com/psf/requests", + ) + + weak_enriched, strong_enriched = enricher.enrich_components([weak_component, strong_component]) + + assert weak_enriched.scorecard is not None + assert weak_enriched.scorecard.statuses == (ScorecardStatus.REPOSITORY_UNMAPPED,) + assert strong_enriched.scorecard is not None + assert strong_enriched.scorecard.statuses == (ScorecardStatus.SCORECARD_AVAILABLE,) + assert client.calls == [("github.com", "psf", "requests")] + + +class FakeScorecardClient: + def __init__( + self, + *, + responses: dict[tuple[str, str, str], ScorecardProjectResult] | None = None, + errors: dict[tuple[str, str, str], Exception] | None = None, + ) -> None: + self._responses = responses or {} + self._errors = errors or {} + self.calls: list[tuple[str, str, str]] = [] + + def fetch_project(self, platform: str, owner: str, repo: str) -> ScorecardProjectResult: + key = (platform, owner, repo) + self.calls.append(key) + if key in self._errors: + raise self._errors[key] + return self._responses[key] diff --git a/tools/sbom-diff-and-risk/tests/test_scorecard_reporting.py b/tools/sbom-diff-and-risk/tests/test_scorecard_reporting.py new file mode 100644 index 0000000..f605d2d --- /dev/null +++ b/tools/sbom-diff-and-risk/tests/test_scorecard_reporting.py @@ -0,0 +1,285 @@ +from __future__ import annotations + +import json +from pathlib import Path + +from sbom_diff_risk.diffing import component_key +from sbom_diff_risk.models import ( + CompareReport, + Component, + ComponentChange, + ReportComponents, + ReportEnrichmentMetadata, + ReportMetadata, + ReportSummary, + RepositoryMapping, + ScorecardCheck, + ScorecardEvidence, + ScorecardStatus, + RiskBucket, + RiskFinding, +) +from sbom_diff_risk.policy_evaluator import evaluate_policy +from sbom_diff_risk.policy_models import PolicyConfig +from sbom_diff_risk.report_json import render_report_json +from sbom_diff_risk.report_md import render_report_markdown +from sbom_diff_risk.report_sarif import render_report_sarif, sarif_rule_id_for_policy_violation +from sbom_diff_risk.risk import summarize_risks + + +def test_scorecard_report_json_matches_golden() -> None: + report, _, _ = _build_sample_scorecard_report() + + rendered = render_report_json(report) + expected = _read_example("sample-scorecard-report.json") + + assert rendered == expected + + +def test_scorecard_report_markdown_matches_golden() -> None: + report, _, _ = _build_sample_scorecard_report() + + rendered = render_report_markdown(report) + expected = _read_example("sample-scorecard-report.md") + + assert rendered == expected + + +def test_scorecard_sarif_matches_golden() -> None: + report, before_path, after_path = _build_sample_scorecard_report() + project_root = Path(__file__).resolve().parents[1] + + rendered = render_report_sarif(report, before_path=before_path, after_path=after_path, base_dir=project_root) + expected = _read_example("sample-scorecard-report.sarif") + + assert _normalize_sarif_snapshot(rendered) == _normalize_sarif_snapshot(expected) + + +def test_scorecard_sarif_rule_id_is_stable() -> None: + assert sarif_rule_id_for_policy_violation("scorecard_below_threshold") == "sdr.policy_violation.scorecard_below_threshold" + + +def test_scorecard_only_evidence_does_not_become_sarif_alert_without_policy_gate() -> None: + component = Component( + name="requests", + version="2.32.0", + ecosystem="pypi", + scorecard=_scorecard( + canonical_name="github.com/psf/requests", + score=4.5, + status=ScorecardStatus.SCORECARD_AVAILABLE, + ), + ) + report = CompareReport( + summary=ReportSummary(added=1, removed=0, changed=0, risk_counts=summarize_risks([])), + components=ReportComponents(added=[component], removed=[], changed=[]), + risks=[], + metadata=ReportMetadata( + before_format="requirements-txt", + after_format="requirements-txt", + generated_at=None, + strict=False, + stub=False, + policy_evaluation=evaluate_policy( + PolicyConfig(version=3, minimum_scorecard_score=7.0), + policy_path="policy.yml", + added=[component], + changed=[], + findings=[], + ), + enrichment=ReportEnrichmentMetadata( + mode="opt_in_scorecard", + scorecard_enabled=True, + scorecard_timeout_seconds=3.0, + scorecard_network_access_performed=True, + network_access_performed=True, + scorecard_candidate_components=1, + scorecard_supported_components=1, + scorecard_status_counts={"scorecard_available": 1}, + ), + ), + notes=["OpenSSF Scorecard enrichment was requested explicitly."], + ) + project_root = Path(__file__).resolve().parents[1] + examples = project_root / "examples" + + payload = json.loads( + render_report_sarif(report, before_path=examples / "requirements_before.txt", after_path=examples / "requirements_after.txt", base_dir=project_root) + ) + + assert payload["runs"][0]["results"] == [] + + +def _build_sample_scorecard_report() -> tuple[CompareReport, Path, Path]: + project_root = Path(__file__).resolve().parents[1] + examples = project_root / "examples" + before_path = examples / "requirements_before.txt" + after_path = examples / "requirements_after.txt" + + requests = Component( + name="requests", + version="2.32.0", + ecosystem="pypi", + purl="pkg:pypi/requests@2.32.0", + source_url="https://github.com/psf/requests", + scorecard=_scorecard( + canonical_name="github.com/psf/requests", + score=6.0, + status=ScorecardStatus.SCORECARD_AVAILABLE, + ), + ) + urllib3 = Component( + name="urllib3", + version="2.2.1", + ecosystem="pypi", + purl="pkg:pypi/urllib3@2.2.1", + source_url="https://pypi.org/project/urllib3/2.2.1/", + scorecard=_scorecard( + canonical_name=None, + score=None, + status=ScorecardStatus.REPOSITORY_UNMAPPED, + ), + ) + certifi_before = Component( + name="certifi", + version="2025.1.0", + ecosystem="pypi", + purl="pkg:pypi/certifi@2025.1.0", + ) + certifi_after = Component( + name="certifi", + version="2026.1.1", + ecosystem="pypi", + purl="pkg:pypi/certifi@2026.1.1", + source_url="https://github.com/certifi/python-certifi", + scorecard=_scorecard( + canonical_name="github.com/certifi/python-certifi", + score=8.4, + status=ScorecardStatus.SCORECARD_AVAILABLE, + ), + ) + certifi_change = ComponentChange( + key=component_key(certifi_after), + before=certifi_before, + after=certifi_after, + classification="version_changed", + ) + + risks = [ + RiskFinding( + bucket=RiskBucket.NEW_PACKAGE, + component_key=component_key(requests), + component=requests, + rationale="Component was not present in the before input.", + ), + RiskFinding( + bucket=RiskBucket.NEW_PACKAGE, + component_key=component_key(urllib3), + component=urllib3, + rationale="Component was not present in the before input.", + ), + ] + policy = PolicyConfig( + version=3, + warn_on=("scorecard_below_threshold",), + minimum_scorecard_score=7.0, + ) + policy_evaluation = evaluate_policy( + policy, + policy_path="examples/policy-scorecard-minimal.yml", + added=[requests, urllib3], + changed=[certifi_change], + findings=risks, + ) + report = CompareReport( + summary=ReportSummary( + added=2, + removed=0, + changed=1, + risk_counts=summarize_risks(risks), + ), + components=ReportComponents( + added=[requests, urllib3], + removed=[], + changed=[certifi_change], + ), + risks=risks, + metadata=ReportMetadata( + before_format="requirements-txt", + after_format="requirements-txt", + generated_at=None, + strict=False, + stub=False, + policy_evaluation=policy_evaluation, + enrichment=ReportEnrichmentMetadata( + mode="opt_in_scorecard", + scorecard_enabled=True, + scorecard_timeout_seconds=3.0, + scorecard_network_access_performed=True, + network_access_performed=True, + scorecard_candidate_components=3, + scorecard_supported_components=2, + scorecard_status_counts={ + "repository_unmapped": 1, + "scorecard_available": 2, + }, + ), + ), + notes=[ + "This tool uses heuristic risk classification.", + "OpenSSF Scorecard enrichment was requested explicitly.", + ], + ) + return report, before_path, after_path + + +def _scorecard( + *, + canonical_name: str | None, + score: float | None, + status: ScorecardStatus, +) -> ScorecardEvidence: + repository = None + note = None + checks: tuple[ScorecardCheck, ...] = () + if canonical_name is not None: + _, owner, repo = canonical_name.split("/", 2) + repository = RepositoryMapping( + platform="github.com", + owner=owner, + repo=repo, + canonical_name=canonical_name, + repository_url=f"https://{canonical_name}", + source="component.source_url", + ) + checks = ( + ScorecardCheck(name="Maintained", score=10, reason="Project is active."), + ScorecardCheck(name="Binary-Artifacts", score=8, reason="No unexpected artifacts were found."), + ) + else: + note = "No high-confidence source repository mapping was available from explicit component metadata." + + return ScorecardEvidence( + provider="openssf-scorecard", + requested=True, + repository=repository, + statuses=(status,), + score=score, + date="2026-04-10T00:00:00Z" if score is not None else None, + scorecard_version="5.0.0" if score is not None else None, + scorecard_commit="def456" if score is not None else None, + repository_commit="abc123" if score is not None else None, + checks=checks, + note=note, + ) + + +def _read_example(name: str) -> str: + examples = Path(__file__).resolve().parents[1] / "examples" + return (examples / name).read_text(encoding="utf-8") + + +def _normalize_sarif_snapshot(content: str) -> str: + payload = json.loads(content) + payload["runs"][0]["originalUriBaseIds"]["%SRCROOT%"]["uri"] = "file:///__PROJECT_ROOT__/" + return json.dumps(payload, indent=2) + "\n"