From 288268e72abefc13f7caceda38146b103b897efc Mon Sep 17 00:00:00 2001 From: stacknil Date: Wed, 15 Apr 2026 00:04:48 +0800 Subject: [PATCH 1/6] Add policy schema and stable enforcement exit codes --- tools/sbom-diff-and-risk/README.md | 36 ++++ .../sbom-diff-and-risk/docs/policy-schema.md | 56 +++++++ .../examples/policy-minimal.yml | 5 + .../examples/policy-strict.yml | 15 ++ .../examples/sample-report.json | 15 +- .../examples/sample-requirements-report.json | 15 +- tools/sbom-diff-and-risk/pyproject.toml | 1 + .../src/sbom_diff_risk/cli.py | 16 +- .../src/sbom_diff_risk/errors.py | 4 + .../src/sbom_diff_risk/models.py | 6 +- .../src/sbom_diff_risk/policy_evaluator.py | 138 ++++++++++++++++ .../src/sbom_diff_risk/policy_models.py | 52 ++++++ .../src/sbom_diff_risk/policy_parser.py | 156 ++++++++++++++++++ .../src/sbom_diff_risk/report_json.py | 57 +++++++ .../tests/test_cli_exit_codes.py | 88 ++++++++++ tools/sbom-diff-and-risk/tests/test_policy.py | 140 ++++++++++++++++ 16 files changed, 796 insertions(+), 4 deletions(-) create mode 100644 tools/sbom-diff-and-risk/docs/policy-schema.md create mode 100644 tools/sbom-diff-and-risk/examples/policy-minimal.yml create mode 100644 tools/sbom-diff-and-risk/examples/policy-strict.yml create mode 100644 tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_evaluator.py create mode 100644 tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_models.py create mode 100644 tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_parser.py create mode 100644 tools/sbom-diff-and-risk/tests/test_cli_exit_codes.py create mode 100644 tools/sbom-diff-and-risk/tests/test_policy.py diff --git a/tools/sbom-diff-and-risk/README.md b/tools/sbom-diff-and-risk/README.md index f76b40b..17ebaa4 100644 --- a/tools/sbom-diff-and-risk/README.md +++ b/tools/sbom-diff-and-risk/README.md @@ -131,6 +131,9 @@ sbom-diff-risk compare \ - `--after-format cyclonedx-json|spdx-json|requirements-txt|pyproject-toml` - `--out-json path` - `--out-md path` +- `--policy path` +- `--fail-on rule[,rule...]` +- `--warn-on rule[,rule...]` - `--strict` - `--enrich-pypi` - `--source-allowlist pypi.org,files.pythonhosted.org,github.com` @@ -142,10 +145,42 @@ sbom-diff-risk compare \ The [`examples/`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples) directory includes: - before/after inputs for CycloneDX JSON, SPDX JSON, `requirements.txt`, and `pyproject.toml` +- example policies at `examples/policy-minimal.yml` and `examples/policy-strict.yml` - a sample CycloneDX-based JSON report at [`sample-report.json`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-report.json) - a sample CycloneDX-based Markdown report at [`sample-report.md`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-report.md) - requirements-based sample reports at [`sample-requirements-report.json`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-requirements-report.json) and [`sample-requirements-report.md`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-requirements-report.md) +## Enforcement + +Policy enforcement is optional and deterministic. Exit codes are stable: + +- `0` = success / no blocking violations +- `1` = blocking policy violations +- `2` = usage, parse, policy, or runtime error + +Minimal policy enforcement example: + +```bash +sbom-diff-risk compare \ + --before examples/requirements_before.txt \ + --after examples/requirements_after.txt \ + --policy examples/policy-minimal.yml \ + --out-json outputs/report.json \ + --out-md outputs/report.md +``` + +Ad hoc enforcement without a policy file: + +```bash +sbom-diff-risk compare \ + --before examples/cdx_before.json \ + --after examples/cdx_after.json \ + --fail-on suspicious_source,unknown_license \ + --warn-on new_package \ + --out-json outputs/report.json \ + --out-md outputs/report.md +``` + ## Limitations - v0.1 is local-file based only. @@ -158,6 +193,7 @@ The [`examples/`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff- - `pyproject.toml` intentionally does not support tool-specific layouts such as Poetry, Hatch, or PDM sections in v0.1. - Risk buckets are heuristics, not security verdicts. - Runtime-generated `outputs/` artifacts are ignored; tracked examples live in `examples/`. +- Policy files are YAML-only in v0.1 and unknown rule ids fail closed. ## Current Status diff --git a/tools/sbom-diff-and-risk/docs/policy-schema.md b/tools/sbom-diff-and-risk/docs/policy-schema.md new file mode 100644 index 0000000..3c92457 --- /dev/null +++ b/tools/sbom-diff-and-risk/docs/policy-schema.md @@ -0,0 +1,56 @@ +# Policy schema + +`sbom-diff-and-risk` supports a YAML-only policy schema in v1. + +The schema is intentionally conservative and fail-closed: + +- unknown rule ids are rejected +- unknown top-level keys are rejected +- invalid types are rejected +- only schema version `1` is supported + +## Fields + +- `version: 1` +- `block_on: [rule_id, ...]` +- `warn_on: [rule_id, ...]` +- `max_added_packages: int` +- `allow_sources: [host, ...]` +- `ignore_rules: [rule_id, ...]` + +## Supported rule ids + +- `new_package` +- `major_upgrade` +- `version_change_unclassified` +- `unknown_license` +- `suspicious_source` +- `stale_package` +- `max_added_packages` +- `allow_sources` + +## Semantics + +- `block_on` turns matching rule ids into blocking violations. +- `warn_on` turns matching rule ids into warnings. +- If a rule is present in both `block_on` and `warn_on`, block wins. +- `max_added_packages` enforces a deterministic threshold on the added component count. +- `allow_sources` enforces exact host matches against `source_url` hosts for added and changed components. +- `ignore_rules` suppresses matching rule ids entirely. + +## Example + +```yaml +version: 1 +block_on: + - unknown_license + - stale_package +warn_on: + - new_package +max_added_packages: 2 +allow_sources: + - pypi.org + - files.pythonhosted.org +ignore_rules: + - major_upgrade +``` diff --git a/tools/sbom-diff-and-risk/examples/policy-minimal.yml b/tools/sbom-diff-and-risk/examples/policy-minimal.yml new file mode 100644 index 0000000..0a858dd --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/policy-minimal.yml @@ -0,0 +1,5 @@ +version: 1 +block_on: + - unknown_license +warn_on: + - new_package diff --git a/tools/sbom-diff-and-risk/examples/policy-strict.yml b/tools/sbom-diff-and-risk/examples/policy-strict.yml new file mode 100644 index 0000000..9ebe01c --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/policy-strict.yml @@ -0,0 +1,15 @@ +version: 1 +block_on: + - unknown_license + - suspicious_source + - stale_package + - max_added_packages + - allow_sources +warn_on: + - new_package + - major_upgrade +max_added_packages: 0 +allow_sources: + - pypi.org + - files.pythonhosted.org + - github.com diff --git a/tools/sbom-diff-and-risk/examples/sample-report.json b/tools/sbom-diff-and-risk/examples/sample-report.json index aa86a8e..e680946 100644 --- a/tools/sbom-diff-and-risk/examples/sample-report.json +++ b/tools/sbom-diff-and-risk/examples/sample-report.json @@ -305,7 +305,20 @@ "after_format": "cyclonedx-json", "generated_at": null, "strict": false, - "stub": false + "stub": false, + "policy_evaluation": { + "applied": false, + "policy_path": null, + "effective_policy": null, + "blocking_violations": [], + "warning_violations": [], + "totals": { + "blocking": 0, + "warning": 0, + "ignored_checks": 0 + }, + "exit_code": 0 + } }, "notes": [ "This tool uses heuristic risk classification.", diff --git a/tools/sbom-diff-and-risk/examples/sample-requirements-report.json b/tools/sbom-diff-and-risk/examples/sample-requirements-report.json index ffe8e40..6664bfd 100644 --- a/tools/sbom-diff-and-risk/examples/sample-requirements-report.json +++ b/tools/sbom-diff-and-risk/examples/sample-requirements-report.json @@ -241,7 +241,20 @@ "after_format": "requirements-txt", "generated_at": null, "strict": false, - "stub": false + "stub": false, + "policy_evaluation": { + "applied": false, + "policy_path": null, + "effective_policy": null, + "blocking_violations": [], + "warning_violations": [], + "totals": { + "blocking": 0, + "warning": 0, + "ignored_checks": 0 + }, + "exit_code": 0 + } }, "notes": [ "This tool uses heuristic risk classification.", diff --git a/tools/sbom-diff-and-risk/pyproject.toml b/tools/sbom-diff-and-risk/pyproject.toml index 5bc3bca..1a3e167 100644 --- a/tools/sbom-diff-and-risk/pyproject.toml +++ b/tools/sbom-diff-and-risk/pyproject.toml @@ -14,6 +14,7 @@ authors = [ ] dependencies = [ "packaging>=24.0", + "PyYAML>=6.0", ] [project.optional-dependencies] diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/cli.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/cli.py index a1ca3e5..64714b9 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/cli.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/cli.py @@ -8,6 +8,8 @@ from .diffing import diff_components from .models import CompareReport, ReportComponents, ReportMetadata, ReportSummary from .normalize import SUPPORTED_FORMATS, normalize_input +from .policy_evaluator import evaluate_policy +from .policy_parser import build_policy from .report_json import render_report_json from .report_md import render_report_markdown from .risk import evaluate_risks, summarize_risks @@ -46,6 +48,9 @@ def build_parser() -> argparse.ArgumentParser: ) compare.add_argument("--out-json", type=Path, default=None, help="Write a JSON report to this path.") compare.add_argument("--out-md", type=Path, default=None, help="Write a Markdown report to this path.") + compare.add_argument("--policy", type=Path, default=None, help="Path to a YAML policy file.") + compare.add_argument("--fail-on", default=None, help="Comma-separated policy rule ids that should block.") + compare.add_argument("--warn-on", default=None, help="Comma-separated policy rule ids that should warn.") compare.add_argument("--strict", action="store_true", help="Treat scaffold notes as errors.") compare.add_argument( "--enrich-pypi", @@ -90,10 +95,18 @@ def run_compare(args: argparse.Namespace) -> int: after_declared = _resolve_declared_format(args.format, args.after_format) before_format, before_components, before_notes = normalize_input(before_path, before_declared) after_format, after_components, after_notes = normalize_input(after_path, after_declared) + policy, policy_path = build_policy(policy_path=args.policy, fail_on=args.fail_on, warn_on=args.warn_on) added, removed, changed = diff_components(before_components, after_components) allowlist = [entry.strip() for entry in args.source_allowlist.split(",") if entry.strip()] risks = evaluate_risks(added, changed, allowlist=allowlist) + policy_evaluation = evaluate_policy( + policy, + policy_path=policy_path, + added=added, + changed=changed, + findings=risks, + ) notes = [ "This tool uses heuristic risk classification.", @@ -117,6 +130,7 @@ def run_compare(args: argparse.Namespace) -> int: generated_at=None, strict=args.strict, stub=False, + policy_evaluation=policy_evaluation, ), notes=notes, ) @@ -129,7 +143,7 @@ def run_compare(args: argparse.Namespace) -> int: if args.out_md is not None: _write_text(args.out_md, render_report_markdown(report)) - return 0 + return policy_evaluation.exit_code def _resolve_declared_format(global_format: str, explicit_format: str | None) -> str | None: diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/errors.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/errors.py index 7e9e7a3..6f8f189 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/errors.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/errors.py @@ -3,3 +3,7 @@ class ParseError(ValueError): """Raised when an input file cannot be parsed into normalized components.""" + + +class PolicyError(ValueError): + """Raised when a policy file or policy override is invalid.""" diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/models.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/models.py index 26a8e0e..ca97cf6 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/models.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/models.py @@ -2,7 +2,10 @@ from dataclasses import dataclass, field from enum import StrEnum -from typing import Any +from typing import TYPE_CHECKING, Any + +if TYPE_CHECKING: + from .policy_models import PolicyEvaluation class RiskBucket(StrEnum): @@ -67,6 +70,7 @@ class ReportMetadata: generated_at: str | None = None strict: bool = False stub: bool = True + policy_evaluation: PolicyEvaluation | None = None @dataclass(slots=True) diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_evaluator.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_evaluator.py new file mode 100644 index 0000000..86050ce --- /dev/null +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_evaluator.py @@ -0,0 +1,138 @@ +from __future__ import annotations + +from urllib.parse import urlparse + +from .diffing import component_key +from .models import Component, ComponentChange, RiskBucket, RiskFinding +from .policy_models import PolicyConfig, PolicyEvaluation, PolicyLevel, PolicyViolation + + +def evaluate_policy( + policy: PolicyConfig | None, + *, + policy_path: str | None, + added: list[Component], + changed: list[ComponentChange], + findings: list[RiskFinding], +) -> PolicyEvaluation: + if policy is None: + return PolicyEvaluation(applied=False, policy_path=None, effective_policy=None, exit_code=0) + + blocking_violations: list[PolicyViolation] = [] + warning_violations: list[PolicyViolation] = [] + ignored_checks = 0 + + for finding in findings: + rule_id = finding_rule_id(finding) + if rule_id in policy.ignore_rules: + ignored_checks += 1 + continue + + level = _severity_for_rule(policy, rule_id) + if level is None: + continue + + violation = PolicyViolation( + rule_id=rule_id, + level=level, + message=finding.rationale, + component_key=finding.component_key, + component_name=finding.component.name, + finding_bucket=finding.bucket.value, + ) + _append_violation(violation, blocking_violations, warning_violations) + + if policy.max_added_packages is not None and len(added) > policy.max_added_packages: + rule_id = "max_added_packages" + if rule_id in policy.ignore_rules: + ignored_checks += 1 + else: + level = _severity_for_rule(policy, rule_id, default=PolicyLevel.BLOCK) + if level is not None: + violation = PolicyViolation( + rule_id=rule_id, + level=level, + message=( + f"Added package count {len(added)} exceeds max_added_packages=" + f"{policy.max_added_packages}." + ), + ) + _append_violation(violation, blocking_violations, warning_violations) + + if policy.allow_sources: + for component in _components_for_source_policy(added, changed): + host = _source_host(component.source_url) + if host in policy.allow_sources: + continue + + rule_id = "allow_sources" + if rule_id in policy.ignore_rules: + ignored_checks += 1 + continue + + level = _severity_for_rule(policy, rule_id, default=PolicyLevel.BLOCK) + if level is None: + continue + + violation = PolicyViolation( + rule_id=rule_id, + level=level, + message=f"Source host {host} is not present in allow_sources.", + component_key=component_key(component), + component_name=component.name, + ) + _append_violation(violation, blocking_violations, warning_violations) + + exit_code = 1 if blocking_violations else 0 + return PolicyEvaluation( + applied=True, + policy_path=policy_path, + effective_policy=policy, + blocking_violations=blocking_violations, + warning_violations=warning_violations, + ignored_checks=ignored_checks, + exit_code=exit_code, + ) + + +def finding_rule_id(finding: RiskFinding) -> str: + if finding.bucket in {RiskBucket.STALE_PACKAGE, RiskBucket.NOT_EVALUATED}: + return "stale_package" + return finding.bucket.value + + +def _severity_for_rule( + policy: PolicyConfig, + rule_id: str, + *, + default: PolicyLevel | None = None, +) -> PolicyLevel | None: + if rule_id in policy.block_on: + return PolicyLevel.BLOCK + if rule_id in policy.warn_on: + return PolicyLevel.WARN + return default + + +def _append_violation( + violation: PolicyViolation, + blocking_violations: list[PolicyViolation], + warning_violations: list[PolicyViolation], +) -> None: + if violation.level is PolicyLevel.BLOCK: + blocking_violations.append(violation) + else: + warning_violations.append(violation) + + +def _components_for_source_policy(added: list[Component], changed: list[ComponentChange]) -> list[Component]: + components = list(added) + components.extend(change.after for change in changed) + return components + + +def _source_host(source_url: str | None) -> str | None: + if not source_url: + return None + host = (urlparse(source_url).hostname or "").strip().lower() + return host or None diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_models.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_models.py new file mode 100644 index 0000000..20f7d4c --- /dev/null +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_models.py @@ -0,0 +1,52 @@ +from __future__ import annotations + +from dataclasses import dataclass, field +from enum import StrEnum + + +class PolicyLevel(StrEnum): + BLOCK = "block" + WARN = "warn" + + +SUPPORTED_POLICY_RULE_IDS = ( + "new_package", + "major_upgrade", + "version_change_unclassified", + "unknown_license", + "suspicious_source", + "stale_package", + "max_added_packages", + "allow_sources", +) + + +@dataclass(slots=True, frozen=True) +class PolicyConfig: + version: int + block_on: tuple[str, ...] = () + warn_on: tuple[str, ...] = () + max_added_packages: int | None = None + allow_sources: tuple[str, ...] = () + ignore_rules: tuple[str, ...] = () + + +@dataclass(slots=True) +class PolicyViolation: + rule_id: str + level: PolicyLevel + message: str + component_key: str | None = None + component_name: str | None = None + finding_bucket: str | None = None + + +@dataclass(slots=True) +class PolicyEvaluation: + applied: bool + policy_path: str | None = None + effective_policy: PolicyConfig | None = None + blocking_violations: list[PolicyViolation] = field(default_factory=list) + warning_violations: list[PolicyViolation] = field(default_factory=list) + ignored_checks: int = 0 + exit_code: int = 0 diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_parser.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_parser.py new file mode 100644 index 0000000..6b417bf --- /dev/null +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_parser.py @@ -0,0 +1,156 @@ +from __future__ import annotations + +from pathlib import Path +from typing import Iterable + +import yaml + +from .errors import PolicyError +from .policy_models import PolicyConfig, SUPPORTED_POLICY_RULE_IDS + +_SUPPORTED_POLICY_KEYS = { + "version", + "block_on", + "warn_on", + "max_added_packages", + "allow_sources", + "ignore_rules", +} + + +def load_policy(path: Path) -> PolicyConfig: + if not path.is_file(): + raise FileNotFoundError(f"policy file does not exist: {path}") + + try: + payload = yaml.safe_load(path.read_text(encoding="utf-8")) + except yaml.YAMLError as exc: + raise PolicyError(f"Malformed YAML policy in {path}: {exc}.") from exc + + if not isinstance(payload, dict): + raise PolicyError(f"Invalid policy schema in {path}: top-level YAML value must be a mapping.") + + unknown_keys = sorted(set(payload) - _SUPPORTED_POLICY_KEYS) + if unknown_keys: + raise PolicyError(f"Invalid policy schema in {path}: unsupported keys: {', '.join(unknown_keys)}.") + + version = payload.get("version") + if not isinstance(version, int): + raise PolicyError(f"Invalid policy schema in {path}: version must be an integer.") + if version != 1: + raise PolicyError(f"Invalid policy schema in {path}: only version 1 is supported.") + + block_on = _parse_rule_list(payload.get("block_on", []), f"{path}: block_on") + warn_on = _parse_rule_list(payload.get("warn_on", []), f"{path}: warn_on") + ignore_rules = _parse_rule_list(payload.get("ignore_rules", []), f"{path}: ignore_rules") + + max_added_packages = payload.get("max_added_packages") + if max_added_packages is not None and (not isinstance(max_added_packages, int) or max_added_packages < 0): + raise PolicyError(f"Invalid policy schema in {path}: max_added_packages must be a non-negative integer.") + + allow_sources = _parse_string_list(payload.get("allow_sources", []), f"{path}: allow_sources", lower=True) + + return normalize_policy( + PolicyConfig( + version=version, + block_on=block_on, + warn_on=warn_on, + max_added_packages=max_added_packages, + allow_sources=allow_sources, + ignore_rules=ignore_rules, + ) + ) + + +def build_policy( + *, + policy_path: Path | None = None, + fail_on: str | None = None, + warn_on: str | None = None, +) -> tuple[PolicyConfig | None, str | None]: + base_policy: PolicyConfig | None = None + rendered_path: str | None = None + if policy_path is not None: + base_policy = load_policy(policy_path) + rendered_path = str(policy_path) + + cli_block_on = parse_rule_csv(fail_on, "--fail-on") + cli_warn_on = parse_rule_csv(warn_on, "--warn-on") + if base_policy is None and not cli_block_on and not cli_warn_on: + return None, None + + seed = base_policy or PolicyConfig(version=1) + merged = PolicyConfig( + version=seed.version, + block_on=_merge_strings(seed.block_on, cli_block_on), + warn_on=_merge_strings(seed.warn_on, cli_warn_on), + max_added_packages=seed.max_added_packages, + allow_sources=seed.allow_sources, + ignore_rules=seed.ignore_rules, + ) + return normalize_policy(merged), rendered_path + + +def normalize_policy(policy: PolicyConfig) -> PolicyConfig: + block_on = tuple(dict.fromkeys(policy.block_on)) + warn_on = tuple(rule for rule in dict.fromkeys(policy.warn_on) if rule not in block_on) + ignore_rules = tuple(dict.fromkeys(policy.ignore_rules)) + return PolicyConfig( + version=policy.version, + block_on=block_on, + warn_on=warn_on, + max_added_packages=policy.max_added_packages, + allow_sources=tuple(dict.fromkeys(policy.allow_sources)), + ignore_rules=ignore_rules, + ) + + +def parse_rule_csv(value: str | None, source_name: str) -> tuple[str, ...]: + if value is None: + return () + entries = [entry.strip() for entry in value.split(",")] + parsed = [entry for entry in entries if entry] + if not parsed: + raise PolicyError(f"{source_name} requires at least one rule id.") + return _validate_rule_ids(parsed, source_name) + + +def _parse_rule_list(value: object, context: str) -> tuple[str, ...]: + if value is None: + return () + if not isinstance(value, list): + raise PolicyError(f"Invalid policy schema in {context}: expected a YAML list of rule ids.") + if not all(isinstance(item, str) for item in value): + raise PolicyError(f"Invalid policy schema in {context}: all rule ids must be strings.") + return _validate_rule_ids(value, context) + + +def _parse_string_list(value: object, context: str, *, lower: bool = False) -> tuple[str, ...]: + if value is None: + return () + if not isinstance(value, list): + raise PolicyError(f"Invalid policy schema in {context}: expected a YAML list of strings.") + items: list[str] = [] + for item in value: + if not isinstance(item, str) or not item.strip(): + raise PolicyError(f"Invalid policy schema in {context}: all values must be non-empty strings.") + normalized = item.strip() + if lower: + normalized = normalized.lower() + items.append(normalized) + return tuple(dict.fromkeys(items)) + + +def _validate_rule_ids(rule_ids: Iterable[str], context: str) -> tuple[str, ...]: + normalized = tuple(dict.fromkeys(rule_id.strip() for rule_id in rule_ids if rule_id.strip())) + unknown = sorted(set(normalized) - set(SUPPORTED_POLICY_RULE_IDS)) + if unknown: + raise PolicyError( + f"Unknown rule id(s) in {context}: {', '.join(unknown)}. " + f"Supported rule ids: {', '.join(SUPPORTED_POLICY_RULE_IDS)}." + ) + return normalized + + +def _merge_strings(base: tuple[str, ...], extra: tuple[str, ...]) -> tuple[str, ...]: + return tuple(dict.fromkeys((*base, *extra))) diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_json.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_json.py index a6c5fc8..99ed3b9 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_json.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_json.py @@ -3,6 +3,7 @@ import json from .models import CompareReport, Component, ComponentChange, RiskFinding +from .policy_models import PolicyConfig, PolicyEvaluation, PolicyViolation def render_report_json(report: CompareReport) -> str: @@ -25,6 +26,7 @@ def render_report_json(report: CompareReport) -> str: "generated_at": report.metadata.generated_at, "strict": report.metadata.strict, "stub": report.metadata.stub, + "policy_evaluation": _policy_evaluation_to_dict(report.metadata.policy_evaluation), }, "notes": list(report.notes), } @@ -62,3 +64,58 @@ def _risk_to_dict(finding: RiskFinding) -> dict[str, object]: "component": _component_to_dict(finding.component), "rationale": finding.rationale, } + + +def _policy_evaluation_to_dict(policy_evaluation: PolicyEvaluation | None) -> dict[str, object]: + if policy_evaluation is None: + return { + "applied": False, + "policy_path": None, + "effective_policy": None, + "blocking_violations": [], + "warning_violations": [], + "totals": { + "blocking": 0, + "warning": 0, + "ignored_checks": 0, + }, + "exit_code": 0, + } + + return { + "applied": policy_evaluation.applied, + "policy_path": policy_evaluation.policy_path, + "effective_policy": _policy_config_to_dict(policy_evaluation.effective_policy), + "blocking_violations": [_policy_violation_to_dict(item) for item in policy_evaluation.blocking_violations], + "warning_violations": [_policy_violation_to_dict(item) for item in policy_evaluation.warning_violations], + "totals": { + "blocking": len(policy_evaluation.blocking_violations), + "warning": len(policy_evaluation.warning_violations), + "ignored_checks": policy_evaluation.ignored_checks, + }, + "exit_code": policy_evaluation.exit_code, + } + + +def _policy_config_to_dict(policy: PolicyConfig | None) -> dict[str, object] | None: + if policy is None: + return None + return { + "version": policy.version, + "block_on": list(policy.block_on), + "warn_on": list(policy.warn_on), + "max_added_packages": policy.max_added_packages, + "allow_sources": list(policy.allow_sources), + "ignore_rules": list(policy.ignore_rules), + } + + +def _policy_violation_to_dict(violation: PolicyViolation) -> dict[str, object]: + return { + "rule_id": violation.rule_id, + "level": violation.level.value, + "message": violation.message, + "component_key": violation.component_key, + "component_name": violation.component_name, + "finding_bucket": violation.finding_bucket, + } diff --git a/tools/sbom-diff-and-risk/tests/test_cli_exit_codes.py b/tools/sbom-diff-and-risk/tests/test_cli_exit_codes.py new file mode 100644 index 0000000..851f319 --- /dev/null +++ b/tools/sbom-diff-and-risk/tests/test_cli_exit_codes.py @@ -0,0 +1,88 @@ +from __future__ import annotations + +import subprocess +import sys +from pathlib import Path + + +def test_cli_exit_code_blocking_policy(tmp_path: Path) -> None: + project_root = Path(__file__).resolve().parents[1] + policy_path = project_root / "examples" / "policy-strict.yml" + before = project_root / "examples" / "cdx_before.json" + after = project_root / "examples" / "cdx_after.json" + + result = _run_compare( + project_root, + [ + "--before", + str(before), + "--after", + str(after), + "--policy", + str(policy_path), + "--out-json", + str(tmp_path / "report.json"), + "--out-md", + str(tmp_path / "report.md"), + ], + ) + + assert result.returncode == 1 + + +def test_cli_exit_code_warn_only_policy(tmp_path: Path) -> None: + project_root = Path(__file__).resolve().parents[1] + policy_path = project_root / "examples" / "policy-minimal.yml" + before = project_root / "examples" / "cdx_before.json" + after = project_root / "examples" / "cdx_after.json" + + result = _run_compare( + project_root, + [ + "--before", + str(before), + "--after", + str(after), + "--policy", + str(policy_path), + "--out-json", + str(tmp_path / "report.json"), + "--out-md", + str(tmp_path / "report.md"), + ], + ) + + assert result.returncode == 0 + + +def test_cli_exit_code_invalid_policy_schema(tmp_path: Path) -> None: + project_root = Path(__file__).resolve().parents[1] + before = project_root / "examples" / "cdx_before.json" + after = project_root / "examples" / "cdx_after.json" + policy_path = tmp_path / "policy.yml" + policy_path.write_text("version: 1\nunknown_key: true\n", encoding="utf-8") + + result = _run_compare( + project_root, + [ + "--before", + str(before), + "--after", + str(after), + "--policy", + str(policy_path), + "--out-json", + str(tmp_path / "report.json"), + ], + ) + + assert result.returncode == 2 + + +def _run_compare(project_root: Path, args: list[str]) -> subprocess.CompletedProcess[str]: + return subprocess.run( + [sys.executable, "-m", "sbom_diff_risk.cli", "compare", *args], + cwd=project_root, + text=True, + capture_output=True, + ) diff --git a/tools/sbom-diff-and-risk/tests/test_policy.py b/tools/sbom-diff-and-risk/tests/test_policy.py new file mode 100644 index 0000000..d6aed05 --- /dev/null +++ b/tools/sbom-diff-and-risk/tests/test_policy.py @@ -0,0 +1,140 @@ +from __future__ import annotations + +from pathlib import Path + +import pytest + +from sbom_diff_risk.errors import PolicyError +from sbom_diff_risk.models import Component, ComponentChange, RiskBucket, RiskFinding +from sbom_diff_risk.policy_evaluator import evaluate_policy +from sbom_diff_risk.policy_models import PolicyConfig, PolicyLevel +from sbom_diff_risk.policy_parser import build_policy, load_policy + + +def test_policy_parser_accepts_minimal_policy() -> None: + policy_path = _example_path("policy-minimal.yml") + + policy = load_policy(policy_path) + + assert policy.version == 1 + assert policy.block_on == ("unknown_license",) + assert policy.warn_on == ("new_package",) + + +def test_policy_parser_rejects_unknown_rule_id(tmp_path: Path) -> None: + path = tmp_path / "policy.yml" + path.write_text("version: 1\nblock_on: [made_up]\n", encoding="utf-8") + + with pytest.raises(PolicyError, match="Unknown rule id"): + load_policy(path) + + +def test_policy_parser_rejects_unknown_key(tmp_path: Path) -> None: + path = tmp_path / "policy.yml" + path.write_text("version: 1\nunknown_key: true\n", encoding="utf-8") + + with pytest.raises(PolicyError, match="unsupported keys"): + load_policy(path) + + +def test_policy_parser_rejects_invalid_version(tmp_path: Path) -> None: + path = tmp_path / "policy.yml" + path.write_text("version: 2\n", encoding="utf-8") + + with pytest.raises(PolicyError, match="only version 1"): + load_policy(path) + + +def test_build_policy_merges_cli_rules() -> None: + policy_path = _example_path("policy-minimal.yml") + + policy, policy_path_str = build_policy(policy_path=policy_path, fail_on="suspicious_source", warn_on="new_package") + + assert policy_path_str is not None + assert policy is not None + assert "unknown_license" in policy.block_on + assert "suspicious_source" in policy.block_on + assert "new_package" in policy.warn_on + + +def test_policy_evaluator_blocks_on_finding_bucket() -> None: + policy = PolicyConfig(version=1, block_on=("unknown_license",)) + component = Component(name="requests", version="2.32.0", ecosystem="pypi") + finding = RiskFinding( + bucket=RiskBucket.UNKNOWN_LICENSE, + component_key="purl:pkg:pypi/requests", + component=component, + rationale="License is missing", + ) + + evaluation = evaluate_policy(policy, policy_path="policy.yml", added=[component], changed=[], findings=[finding]) + + assert evaluation.exit_code == 1 + assert len(evaluation.blocking_violations) == 1 + assert evaluation.blocking_violations[0].rule_id == "unknown_license" + assert evaluation.blocking_violations[0].level is PolicyLevel.BLOCK + + +def test_policy_evaluator_warns_on_rule_when_configured() -> None: + policy = PolicyConfig(version=1, warn_on=("new_package",)) + component = Component(name="urllib3", version="2.2.1", ecosystem="pypi") + finding = RiskFinding( + bucket=RiskBucket.NEW_PACKAGE, + component_key="purl:pkg:pypi/urllib3", + component=component, + rationale="New package", + ) + + evaluation = evaluate_policy(policy, policy_path="policy.yml", added=[component], changed=[], findings=[finding]) + + assert evaluation.exit_code == 0 + assert len(evaluation.warning_violations) == 1 + assert evaluation.warning_violations[0].rule_id == "new_package" + + +def test_policy_evaluator_max_added_packages_blocks() -> None: + policy = PolicyConfig(version=1, max_added_packages=0) + added = [ + Component(name="urllib3", version="2.2.1", ecosystem="pypi"), + ] + + evaluation = evaluate_policy(policy, policy_path="policy.yml", added=added, changed=[], findings=[]) + + assert evaluation.exit_code == 1 + assert any(violation.rule_id == "max_added_packages" for violation in evaluation.blocking_violations) + + +def test_policy_evaluator_allow_sources_blocks_unknown_hosts() -> None: + policy = PolicyConfig(version=1, allow_sources=("pypi.org",)) + component = Component( + name="internal-lib", + version="1.0.0", + ecosystem="pypi", + source_url="https://example.com/internal-lib-1.0.0.tar.gz", + ) + + evaluation = evaluate_policy(policy, policy_path="policy.yml", added=[component], changed=[], findings=[]) + + assert evaluation.exit_code == 1 + assert any(violation.rule_id == "allow_sources" for violation in evaluation.blocking_violations) + + +def test_policy_ignore_rules_suppresses_violations() -> None: + policy = PolicyConfig(version=1, block_on=("unknown_license",), ignore_rules=("unknown_license",)) + component = Component(name="requests", version="2.32.0", ecosystem="pypi") + finding = RiskFinding( + bucket=RiskBucket.UNKNOWN_LICENSE, + component_key="purl:pkg:pypi/requests", + component=component, + rationale="License missing", + ) + + evaluation = evaluate_policy(policy, policy_path="policy.yml", added=[component], changed=[], findings=[finding]) + + assert evaluation.exit_code == 0 + assert evaluation.blocking_violations == [] + assert evaluation.ignored_checks == 1 + + +def _example_path(name: str) -> Path: + return Path(__file__).resolve().parents[1] / "examples" / name From 86eb81940c75a1f327fb0c6eb0581b72374bd8a7 Mon Sep 17 00:00:00 2001 From: stacknil Date: Wed, 15 Apr 2026 00:10:25 +0800 Subject: [PATCH 2/6] Add policy-aware reports and SARIF export --- tools/sbom-diff-and-risk/README.md | 29 +- .../examples/sample-policy-fail-report.json | 564 ++++++++++++++++++ .../examples/sample-policy-fail-report.md | 64 ++ .../examples/sample-policy-warn-report.json | 462 ++++++++++++++ .../examples/sample-policy-warn-report.md | 62 ++ .../examples/sample-report.json | 83 +++ .../examples/sample-report.md | 18 + .../examples/sample-requirements-report.json | 83 +++ .../examples/sample-requirements-report.md | 18 + .../examples/sample-sarif.sarif | 300 ++++++++++ .../examples/sarif_after.json | 40 ++ .../examples/sarif_before.json | 27 + .../src/sbom_diff_risk/cli.py | 86 ++- .../src/sbom_diff_risk/policy_evaluator.py | 76 ++- .../src/sbom_diff_risk/policy_models.py | 4 +- .../src/sbom_diff_risk/presentation.py | 150 +++++ .../src/sbom_diff_risk/report_json.py | 65 +- .../src/sbom_diff_risk/report_md.py | 65 ++ .../src/sbom_diff_risk/report_sarif.py | 417 +++++++++++++ .../tests/test_cli_exit_codes.py | 75 ++- .../sbom-diff-and-risk/tests/test_reports.py | 110 +++- tools/sbom-diff-and-risk/tests/test_sarif.py | 221 +++++++ 22 files changed, 2921 insertions(+), 98 deletions(-) create mode 100644 tools/sbom-diff-and-risk/examples/sample-policy-fail-report.json create mode 100644 tools/sbom-diff-and-risk/examples/sample-policy-fail-report.md create mode 100644 tools/sbom-diff-and-risk/examples/sample-policy-warn-report.json create mode 100644 tools/sbom-diff-and-risk/examples/sample-policy-warn-report.md create mode 100644 tools/sbom-diff-and-risk/examples/sample-sarif.sarif create mode 100644 tools/sbom-diff-and-risk/examples/sarif_after.json create mode 100644 tools/sbom-diff-and-risk/examples/sarif_before.json create mode 100644 tools/sbom-diff-and-risk/src/sbom_diff_risk/presentation.py create mode 100644 tools/sbom-diff-and-risk/src/sbom_diff_risk/report_sarif.py create mode 100644 tools/sbom-diff-and-risk/tests/test_sarif.py diff --git a/tools/sbom-diff-and-risk/README.md b/tools/sbom-diff-and-risk/README.md index 17ebaa4..bb8193c 100644 --- a/tools/sbom-diff-and-risk/README.md +++ b/tools/sbom-diff-and-risk/README.md @@ -68,6 +68,7 @@ Offline `stale_package` evaluation is intentionally deferred. When enrichment is - `report.json` - `report.md` +- `report.sarif` ## Install @@ -131,6 +132,7 @@ sbom-diff-risk compare \ - `--after-format cyclonedx-json|spdx-json|requirements-txt|pyproject-toml` - `--out-json path` - `--out-md path` +- `--out-sarif path` - `--policy path` - `--fail-on rule[,rule...]` - `--warn-on rule[,rule...]` @@ -148,6 +150,9 @@ The [`examples/`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff- - example policies at `examples/policy-minimal.yml` and `examples/policy-strict.yml` - a sample CycloneDX-based JSON report at [`sample-report.json`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-report.json) - a sample CycloneDX-based Markdown report at [`sample-report.md`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-report.md) +- sample policy-warn reports at [`sample-policy-warn-report.json`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.json) and [`sample-policy-warn-report.md`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.md) +- sample policy-fail reports at [`sample-policy-fail-report.json`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.json) and [`sample-policy-fail-report.md`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.md) +- a sample SARIF export at [`sample-sarif.sarif`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-sarif.sarif) - requirements-based sample reports at [`sample-requirements-report.json`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-requirements-report.json) and [`sample-requirements-report.md`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-requirements-report.md) ## Enforcement @@ -181,11 +186,33 @@ sbom-diff-risk compare \ --out-md outputs/report.md ``` +Failed runs still write reports on exit code `1`; stderr prints a concise blocking summary so CI logs are understandable without opening raw JSON. + +## SARIF Export + +SARIF export is intentionally conservative. The current renderer emits a GitHub-compatible SARIF 2.1.0 subset for: + +- `suspicious_source` +- `unknown_license` +- `major_upgrade` +- selected blocking policy results such as `max_added_packages` and `allow_sources` + +It does not turn every diff or informational heuristic into a code scanning alert. + +```bash +sbom-diff-risk compare \ + --before examples/sarif_before.json \ + --after examples/sarif_after.json \ + --policy examples/policy-strict.yml \ + --out-sarif outputs/report.sarif +``` + ## Limitations - v0.1 is local-file based only. - `generated_at` remains `null` to preserve deterministic report output. - `stale_package` is not resolved offline. The report emits `not_evaluated` instead. +- SARIF export intentionally covers only a conservative subset of findings in v0.1. - No vulnerability database integration, CVE matching, or advisory enrichment. - `requirements.txt` support intentionally covers a conservative subset: plain PEP 508 requirement entries, comments, direct URL requirements, and line continuations. - `requirements.txt` intentionally does not support pip include/constraint directives such as `-r`, `-c`, or arbitrary install flags in v0.1. @@ -197,4 +224,4 @@ sbom-diff-risk compare \ ## Current Status -The project now normalizes local CycloneDX JSON, SPDX JSON, `requirements.txt`, and PEP 621 `pyproject.toml` inputs into the shared component model, diffs them deterministically, and generates stable JSON/Markdown reports with golden tests. +The project now normalizes local CycloneDX JSON, SPDX JSON, `requirements.txt`, and PEP 621 `pyproject.toml` inputs into the shared component model, diffs them deterministically, and generates stable JSON/Markdown/SARIF reports with golden tests and optional policy enforcement. diff --git a/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.json b/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.json new file mode 100644 index 0000000..a5b7271 --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.json @@ -0,0 +1,564 @@ +{ + "summary": { + "added": 1, + "removed": 0, + "changed": 1, + "risk_counts": { + "new_package": 1, + "major_upgrade": 0, + "version_change_unclassified": 1, + "unknown_license": 0, + "stale_package": 0, + "suspicious_source": 0, + "not_evaluated": 2 + } + }, + "components": { + "added": [ + { + "name": "urllib3", + "version": "2.2.1", + "ecosystem": "pypi", + "purl": "pkg:pypi/urllib3@2.2.1", + "license_id": "MIT", + "supplier": null, + "source_url": "https://pypi.org/project/urllib3/", + "bom_ref": "pkg:pypi/urllib3@2.2.1", + "raw_type": "library", + "evidence": { + "source_format": "cyclonedx-json", + "component": { + "bom-ref": "pkg:pypi/urllib3@2.2.1", + "type": "library", + "name": "urllib3", + "version": "2.2.1", + "purl": "pkg:pypi/urllib3@2.2.1", + "licenses": [ + { + "license": { + "id": "MIT" + } + } + ], + "externalReferences": [ + { + "type": "website", + "url": "https://pypi.org/project/urllib3/" + } + ] + } + } + } + ], + "removed": [], + "changed": [ + { + "key": "purl:pkg:pypi/requests", + "classification": "version_changed", + "before": { + "name": "requests", + "version": "2.31.0", + "ecosystem": "pypi", + "purl": "pkg:pypi/requests@2.31.0", + "license_id": "Apache-2.0", + "supplier": "Python Software Foundation", + "source_url": "https://pypi.org/project/requests/", + "bom_ref": "pkg:pypi/requests@2.31.0", + "raw_type": "library", + "evidence": { + "source_format": "cyclonedx-json", + "component": { + "bom-ref": "pkg:pypi/requests@2.31.0", + "type": "library", + "name": "requests", + "version": "2.31.0", + "purl": "pkg:pypi/requests@2.31.0", + "supplier": { + "name": "Python Software Foundation" + }, + "licenses": [ + { + "license": { + "id": "Apache-2.0" + } + } + ], + "externalReferences": [ + { + "type": "website", + "url": "https://pypi.org/project/requests/" + }, + { + "type": "vcs", + "url": "https://github.com/psf/requests" + } + ] + } + } + }, + "after": { + "name": "requests", + "version": "2.32.0", + "ecosystem": "pypi", + "purl": "pkg:pypi/requests@2.32.0", + "license_id": "Apache-2.0", + "supplier": "Python Software Foundation", + "source_url": "https://pypi.org/project/requests/", + "bom_ref": "pkg:pypi/requests@2.32.0", + "raw_type": "library", + "evidence": { + "source_format": "cyclonedx-json", + "component": { + "bom-ref": "pkg:pypi/requests@2.32.0", + "type": "library", + "name": "requests", + "version": "2.32.0", + "purl": "pkg:pypi/requests@2.32.0", + "supplier": { + "name": "Python Software Foundation" + }, + "licenses": [ + { + "license": { + "id": "Apache-2.0" + } + } + ], + "externalReferences": [ + { + "type": "website", + "url": "https://pypi.org/project/requests/" + } + ] + } + } + } + } + ] + }, + "risks": [ + { + "bucket": "new_package", + "component_key": "purl:pkg:pypi/urllib3", + "component": { + "name": "urllib3", + "version": "2.2.1", + "ecosystem": "pypi", + "purl": "pkg:pypi/urllib3@2.2.1", + "license_id": "MIT", + "supplier": null, + "source_url": "https://pypi.org/project/urllib3/", + "bom_ref": "pkg:pypi/urllib3@2.2.1", + "raw_type": "library", + "evidence": { + "source_format": "cyclonedx-json", + "component": { + "bom-ref": "pkg:pypi/urllib3@2.2.1", + "type": "library", + "name": "urllib3", + "version": "2.2.1", + "purl": "pkg:pypi/urllib3@2.2.1", + "licenses": [ + { + "license": { + "id": "MIT" + } + } + ], + "externalReferences": [ + { + "type": "website", + "url": "https://pypi.org/project/urllib3/" + } + ] + } + } + }, + "rationale": "Component was not present in the before input." + }, + { + "bucket": "not_evaluated", + "component_key": "purl:pkg:pypi/requests", + "component": { + "name": "requests", + "version": "2.32.0", + "ecosystem": "pypi", + "purl": "pkg:pypi/requests@2.32.0", + "license_id": "Apache-2.0", + "supplier": "Python Software Foundation", + "source_url": "https://pypi.org/project/requests/", + "bom_ref": "pkg:pypi/requests@2.32.0", + "raw_type": "library", + "evidence": { + "source_format": "cyclonedx-json", + "component": { + "bom-ref": "pkg:pypi/requests@2.32.0", + "type": "library", + "name": "requests", + "version": "2.32.0", + "purl": "pkg:pypi/requests@2.32.0", + "supplier": { + "name": "Python Software Foundation" + }, + "licenses": [ + { + "license": { + "id": "Apache-2.0" + } + } + ], + "externalReferences": [ + { + "type": "website", + "url": "https://pypi.org/project/requests/" + } + ] + } + } + }, + "rationale": "stale_package was not evaluated because enrichment mode is disabled." + }, + { + "bucket": "not_evaluated", + "component_key": "purl:pkg:pypi/urllib3", + "component": { + "name": "urllib3", + "version": "2.2.1", + "ecosystem": "pypi", + "purl": "pkg:pypi/urllib3@2.2.1", + "license_id": "MIT", + "supplier": null, + "source_url": "https://pypi.org/project/urllib3/", + "bom_ref": "pkg:pypi/urllib3@2.2.1", + "raw_type": "library", + "evidence": { + "source_format": "cyclonedx-json", + "component": { + "bom-ref": "pkg:pypi/urllib3@2.2.1", + "type": "library", + "name": "urllib3", + "version": "2.2.1", + "purl": "pkg:pypi/urllib3@2.2.1", + "licenses": [ + { + "license": { + "id": "MIT" + } + } + ], + "externalReferences": [ + { + "type": "website", + "url": "https://pypi.org/project/urllib3/" + } + ] + } + } + }, + "rationale": "stale_package was not evaluated because enrichment mode is disabled." + }, + { + "bucket": "version_change_unclassified", + "component_key": "purl:pkg:pypi/requests", + "component": { + "name": "requests", + "version": "2.32.0", + "ecosystem": "pypi", + "purl": "pkg:pypi/requests@2.32.0", + "license_id": "Apache-2.0", + "supplier": "Python Software Foundation", + "source_url": "https://pypi.org/project/requests/", + "bom_ref": "pkg:pypi/requests@2.32.0", + "raw_type": "library", + "evidence": { + "source_format": "cyclonedx-json", + "component": { + "bom-ref": "pkg:pypi/requests@2.32.0", + "type": "library", + "name": "requests", + "version": "2.32.0", + "purl": "pkg:pypi/requests@2.32.0", + "supplier": { + "name": "Python Software Foundation" + }, + "licenses": [ + { + "license": { + "id": "Apache-2.0" + } + } + ], + "externalReferences": [ + { + "type": "website", + "url": "https://pypi.org/project/requests/" + } + ] + } + } + }, + "rationale": "Version changed but did not qualify as a parseable SemVer major upgrade." + } + ], + "policy_evaluation": { + "applied": true, + "policy_path": "examples\\policy-strict.yml", + "effective_policy": { + "version": 1, + "block_on": [ + "unknown_license", + "suspicious_source", + "stale_package", + "max_added_packages", + "allow_sources" + ], + "warn_on": [ + "new_package", + "major_upgrade" + ], + "max_added_packages": 0, + "allow_sources": [ + "pypi.org", + "files.pythonhosted.org", + "github.com" + ], + "ignore_rules": [] + }, + "blocking_violations": [ + { + "rule_id": "max_added_packages", + "level": "block", + "message": "Added package count 1 exceeds max_added_packages=0.", + "component_key": null, + "component_name": null, + "finding_bucket": null, + "suppression_reason": null + }, + { + "rule_id": "stale_package", + "level": "block", + "message": "stale_package was not evaluated because enrichment mode is disabled.", + "component_key": "purl:pkg:pypi/requests", + "component_name": "requests", + "finding_bucket": "not_evaluated", + "suppression_reason": null + }, + { + "rule_id": "stale_package", + "level": "block", + "message": "stale_package was not evaluated because enrichment mode is disabled.", + "component_key": "purl:pkg:pypi/urllib3", + "component_name": "urllib3", + "finding_bucket": "not_evaluated", + "suppression_reason": null + } + ], + "warning_violations": [ + { + "rule_id": "new_package", + "level": "warn", + "message": "Component was not present in the before input.", + "component_key": "purl:pkg:pypi/urllib3", + "component_name": "urllib3", + "finding_bucket": "new_package", + "suppression_reason": null + } + ], + "suppressed_violations": [], + "totals": { + "blocking": 3, + "warning": 1, + "suppressed": 0, + "ignored_checks": 0 + }, + "exit_code": 1 + }, + "blocking_findings": [ + { + "rule_id": "max_added_packages", + "level": "block", + "message": "Added package count 1 exceeds max_added_packages=0.", + "component_key": null, + "component_name": null, + "finding_bucket": null, + "suppression_reason": null + }, + { + "rule_id": "stale_package", + "level": "block", + "message": "stale_package was not evaluated because enrichment mode is disabled.", + "component_key": "purl:pkg:pypi/requests", + "component_name": "requests", + "finding_bucket": "not_evaluated", + "suppression_reason": null + }, + { + "rule_id": "stale_package", + "level": "block", + "message": "stale_package was not evaluated because enrichment mode is disabled.", + "component_key": "purl:pkg:pypi/urllib3", + "component_name": "urllib3", + "finding_bucket": "not_evaluated", + "suppression_reason": null + } + ], + "warning_findings": [ + { + "rule_id": "new_package", + "level": "warn", + "message": "Component was not present in the before input.", + "component_key": "purl:pkg:pypi/urllib3", + "component_name": "urllib3", + "finding_bucket": "new_package", + "suppression_reason": null + } + ], + "suppressed_findings": [], + "rule_catalog": { + "new_package": { + "rule_id": "new_package", + "kind": "risk_finding", + "description": "Component is present only in the after input.", + "finding_buckets": [ + "new_package" + ] + }, + "major_upgrade": { + "rule_id": "major_upgrade", + "kind": "risk_finding", + "description": "Version change is a parseable SemVer major upgrade.", + "finding_buckets": [ + "major_upgrade" + ] + }, + "version_change_unclassified": { + "rule_id": "version_change_unclassified", + "kind": "risk_finding", + "description": "Version changed but could not be classified as a reliable major SemVer upgrade.", + "finding_buckets": [ + "version_change_unclassified" + ] + }, + "unknown_license": { + "rule_id": "unknown_license", + "kind": "risk_finding", + "description": "License metadata is missing, empty, UNKNOWN, or NOASSERTION.", + "finding_buckets": [ + "unknown_license" + ] + }, + "suspicious_source": { + "rule_id": "suspicious_source", + "kind": "risk_finding", + "description": "Source provenance is missing or points to a suspicious scheme, path, or host.", + "finding_buckets": [ + "suspicious_source" + ] + }, + "stale_package": { + "rule_id": "stale_package", + "kind": "risk_finding", + "description": "Staleness check result. Offline mode maps this rule to not_evaluated instead of guessing.", + "finding_buckets": [ + "stale_package", + "not_evaluated" + ] + }, + "max_added_packages": { + "rule_id": "max_added_packages", + "kind": "policy_check", + "description": "Added package count exceeded the configured deterministic threshold.", + "finding_buckets": [] + }, + "allow_sources": { + "rule_id": "allow_sources", + "kind": "policy_check", + "description": "Component source host was not present in the configured allow_sources list.", + "finding_buckets": [] + } + }, + "metadata": { + "before_format": "cyclonedx-json", + "after_format": "cyclonedx-json", + "generated_at": null, + "strict": false, + "stub": false, + "policy_evaluation": { + "applied": true, + "policy_path": "examples\\policy-strict.yml", + "effective_policy": { + "version": 1, + "block_on": [ + "unknown_license", + "suspicious_source", + "stale_package", + "max_added_packages", + "allow_sources" + ], + "warn_on": [ + "new_package", + "major_upgrade" + ], + "max_added_packages": 0, + "allow_sources": [ + "pypi.org", + "files.pythonhosted.org", + "github.com" + ], + "ignore_rules": [] + }, + "blocking_violations": [ + { + "rule_id": "max_added_packages", + "level": "block", + "message": "Added package count 1 exceeds max_added_packages=0.", + "component_key": null, + "component_name": null, + "finding_bucket": null, + "suppression_reason": null + }, + { + "rule_id": "stale_package", + "level": "block", + "message": "stale_package was not evaluated because enrichment mode is disabled.", + "component_key": "purl:pkg:pypi/requests", + "component_name": "requests", + "finding_bucket": "not_evaluated", + "suppression_reason": null + }, + { + "rule_id": "stale_package", + "level": "block", + "message": "stale_package was not evaluated because enrichment mode is disabled.", + "component_key": "purl:pkg:pypi/urllib3", + "component_name": "urllib3", + "finding_bucket": "not_evaluated", + "suppression_reason": null + } + ], + "warning_violations": [ + { + "rule_id": "new_package", + "level": "warn", + "message": "Component was not present in the before input.", + "component_key": "purl:pkg:pypi/urllib3", + "component_name": "urllib3", + "finding_bucket": "new_package", + "suppression_reason": null + } + ], + "suppressed_violations": [], + "totals": { + "blocking": 3, + "warning": 1, + "suppressed": 0, + "ignored_checks": 0 + }, + "exit_code": 1 + } + }, + "notes": [ + "This tool uses heuristic risk classification.", + "No network enrichment was performed." + ] +} diff --git a/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.md b/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.md new file mode 100644 index 0000000..20485a7 --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.md @@ -0,0 +1,64 @@ +# sbom-diff-and-risk report + +## Summary +- Before format: cyclonedx-json +- After format: cyclonedx-json +- Added: 1 +- Removed: 0 +- Version changes: 1 + +## Risk buckets +- new_package: 1 +- major_upgrade: 0 +- version_change_unclassified: 1 +- unknown_license: 0 +- stale_package: 0 +- suspicious_source: 0 +- not_evaluated: 2 + +## Policy summary +- Applied: yes +- Policy path: examples\policy-strict.yml +- Exit code: 1 +- Blocking findings: 3 +- Warnings: 1 +- Suppressed findings: 0 + +## Added components +| name | version | ecosystem | risk buckets | +|------|---------|-----------|--------------| +| urllib3 | 2.2.1 | pypi | new_package, not_evaluated | + +## Removed components +| name | version | ecosystem | +|------|---------|-----------| +| _none_ | | | + +## Version changes +| name | before | after | classification | risk buckets | +|------|--------|-------|----------------|--------------| +| requests | 2.31.0 | 2.32.0 | version_changed | not_evaluated, version_change_unclassified | + +## Risk findings +| bucket | component | version | rationale | +|--------|-----------|---------|-----------| +| new_package | urllib3 | 2.2.1 | Component was not present in the before input. | +| not_evaluated | requests | 2.32.0 | stale_package was not evaluated because enrichment mode is disabled. | +| not_evaluated | urllib3 | 2.2.1 | stale_package was not evaluated because enrichment mode is disabled. | +| version_change_unclassified | requests | 2.32.0 | Version changed but did not qualify as a parseable SemVer major upgrade. | + +## Blocking violations +| rule id | component | level | message | +|---------|-----------|-------|---------| +| max_added_packages | | block | Added package count 1 exceeds max_added_packages=0. | +| stale_package | requests | block | stale_package was not evaluated because enrichment mode is disabled. | +| stale_package | urllib3 | block | stale_package was not evaluated because enrichment mode is disabled. | + +## Warnings +| rule id | component | level | message | +|---------|-----------|-------|---------| +| new_package | urllib3 | warn | Component was not present in the before input. | + +## Notes +- This tool uses heuristic risk classification. +- No network enrichment was performed. diff --git a/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.json b/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.json new file mode 100644 index 0000000..4410c59 --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.json @@ -0,0 +1,462 @@ +{ + "summary": { + "added": 1, + "removed": 0, + "changed": 1, + "risk_counts": { + "new_package": 1, + "major_upgrade": 0, + "version_change_unclassified": 1, + "unknown_license": 0, + "stale_package": 0, + "suspicious_source": 0, + "not_evaluated": 2 + } + }, + "components": { + "added": [ + { + "name": "urllib3", + "version": "2.2.1", + "ecosystem": "pypi", + "purl": "pkg:pypi/urllib3@2.2.1", + "license_id": "MIT", + "supplier": null, + "source_url": "https://pypi.org/project/urllib3/", + "bom_ref": "pkg:pypi/urllib3@2.2.1", + "raw_type": "library", + "evidence": { + "source_format": "cyclonedx-json", + "component": { + "bom-ref": "pkg:pypi/urllib3@2.2.1", + "type": "library", + "name": "urllib3", + "version": "2.2.1", + "purl": "pkg:pypi/urllib3@2.2.1", + "licenses": [ + { + "license": { + "id": "MIT" + } + } + ], + "externalReferences": [ + { + "type": "website", + "url": "https://pypi.org/project/urllib3/" + } + ] + } + } + } + ], + "removed": [], + "changed": [ + { + "key": "purl:pkg:pypi/requests", + "classification": "version_changed", + "before": { + "name": "requests", + "version": "2.31.0", + "ecosystem": "pypi", + "purl": "pkg:pypi/requests@2.31.0", + "license_id": "Apache-2.0", + "supplier": "Python Software Foundation", + "source_url": "https://pypi.org/project/requests/", + "bom_ref": "pkg:pypi/requests@2.31.0", + "raw_type": "library", + "evidence": { + "source_format": "cyclonedx-json", + "component": { + "bom-ref": "pkg:pypi/requests@2.31.0", + "type": "library", + "name": "requests", + "version": "2.31.0", + "purl": "pkg:pypi/requests@2.31.0", + "supplier": { + "name": "Python Software Foundation" + }, + "licenses": [ + { + "license": { + "id": "Apache-2.0" + } + } + ], + "externalReferences": [ + { + "type": "website", + "url": "https://pypi.org/project/requests/" + }, + { + "type": "vcs", + "url": "https://github.com/psf/requests" + } + ] + } + } + }, + "after": { + "name": "requests", + "version": "2.32.0", + "ecosystem": "pypi", + "purl": "pkg:pypi/requests@2.32.0", + "license_id": "Apache-2.0", + "supplier": "Python Software Foundation", + "source_url": "https://pypi.org/project/requests/", + "bom_ref": "pkg:pypi/requests@2.32.0", + "raw_type": "library", + "evidence": { + "source_format": "cyclonedx-json", + "component": { + "bom-ref": "pkg:pypi/requests@2.32.0", + "type": "library", + "name": "requests", + "version": "2.32.0", + "purl": "pkg:pypi/requests@2.32.0", + "supplier": { + "name": "Python Software Foundation" + }, + "licenses": [ + { + "license": { + "id": "Apache-2.0" + } + } + ], + "externalReferences": [ + { + "type": "website", + "url": "https://pypi.org/project/requests/" + } + ] + } + } + } + } + ] + }, + "risks": [ + { + "bucket": "new_package", + "component_key": "purl:pkg:pypi/urllib3", + "component": { + "name": "urllib3", + "version": "2.2.1", + "ecosystem": "pypi", + "purl": "pkg:pypi/urllib3@2.2.1", + "license_id": "MIT", + "supplier": null, + "source_url": "https://pypi.org/project/urllib3/", + "bom_ref": "pkg:pypi/urllib3@2.2.1", + "raw_type": "library", + "evidence": { + "source_format": "cyclonedx-json", + "component": { + "bom-ref": "pkg:pypi/urllib3@2.2.1", + "type": "library", + "name": "urllib3", + "version": "2.2.1", + "purl": "pkg:pypi/urllib3@2.2.1", + "licenses": [ + { + "license": { + "id": "MIT" + } + } + ], + "externalReferences": [ + { + "type": "website", + "url": "https://pypi.org/project/urllib3/" + } + ] + } + } + }, + "rationale": "Component was not present in the before input." + }, + { + "bucket": "not_evaluated", + "component_key": "purl:pkg:pypi/requests", + "component": { + "name": "requests", + "version": "2.32.0", + "ecosystem": "pypi", + "purl": "pkg:pypi/requests@2.32.0", + "license_id": "Apache-2.0", + "supplier": "Python Software Foundation", + "source_url": "https://pypi.org/project/requests/", + "bom_ref": "pkg:pypi/requests@2.32.0", + "raw_type": "library", + "evidence": { + "source_format": "cyclonedx-json", + "component": { + "bom-ref": "pkg:pypi/requests@2.32.0", + "type": "library", + "name": "requests", + "version": "2.32.0", + "purl": "pkg:pypi/requests@2.32.0", + "supplier": { + "name": "Python Software Foundation" + }, + "licenses": [ + { + "license": { + "id": "Apache-2.0" + } + } + ], + "externalReferences": [ + { + "type": "website", + "url": "https://pypi.org/project/requests/" + } + ] + } + } + }, + "rationale": "stale_package was not evaluated because enrichment mode is disabled." + }, + { + "bucket": "not_evaluated", + "component_key": "purl:pkg:pypi/urllib3", + "component": { + "name": "urllib3", + "version": "2.2.1", + "ecosystem": "pypi", + "purl": "pkg:pypi/urllib3@2.2.1", + "license_id": "MIT", + "supplier": null, + "source_url": "https://pypi.org/project/urllib3/", + "bom_ref": "pkg:pypi/urllib3@2.2.1", + "raw_type": "library", + "evidence": { + "source_format": "cyclonedx-json", + "component": { + "bom-ref": "pkg:pypi/urllib3@2.2.1", + "type": "library", + "name": "urllib3", + "version": "2.2.1", + "purl": "pkg:pypi/urllib3@2.2.1", + "licenses": [ + { + "license": { + "id": "MIT" + } + } + ], + "externalReferences": [ + { + "type": "website", + "url": "https://pypi.org/project/urllib3/" + } + ] + } + } + }, + "rationale": "stale_package was not evaluated because enrichment mode is disabled." + }, + { + "bucket": "version_change_unclassified", + "component_key": "purl:pkg:pypi/requests", + "component": { + "name": "requests", + "version": "2.32.0", + "ecosystem": "pypi", + "purl": "pkg:pypi/requests@2.32.0", + "license_id": "Apache-2.0", + "supplier": "Python Software Foundation", + "source_url": "https://pypi.org/project/requests/", + "bom_ref": "pkg:pypi/requests@2.32.0", + "raw_type": "library", + "evidence": { + "source_format": "cyclonedx-json", + "component": { + "bom-ref": "pkg:pypi/requests@2.32.0", + "type": "library", + "name": "requests", + "version": "2.32.0", + "purl": "pkg:pypi/requests@2.32.0", + "supplier": { + "name": "Python Software Foundation" + }, + "licenses": [ + { + "license": { + "id": "Apache-2.0" + } + } + ], + "externalReferences": [ + { + "type": "website", + "url": "https://pypi.org/project/requests/" + } + ] + } + } + }, + "rationale": "Version changed but did not qualify as a parseable SemVer major upgrade." + } + ], + "policy_evaluation": { + "applied": true, + "policy_path": "examples\\policy-minimal.yml", + "effective_policy": { + "version": 1, + "block_on": [ + "unknown_license" + ], + "warn_on": [ + "new_package" + ], + "max_added_packages": null, + "allow_sources": [], + "ignore_rules": [] + }, + "blocking_violations": [], + "warning_violations": [ + { + "rule_id": "new_package", + "level": "warn", + "message": "Component was not present in the before input.", + "component_key": "purl:pkg:pypi/urllib3", + "component_name": "urllib3", + "finding_bucket": "new_package", + "suppression_reason": null + } + ], + "suppressed_violations": [], + "totals": { + "blocking": 0, + "warning": 1, + "suppressed": 0, + "ignored_checks": 0 + }, + "exit_code": 0 + }, + "blocking_findings": [], + "warning_findings": [ + { + "rule_id": "new_package", + "level": "warn", + "message": "Component was not present in the before input.", + "component_key": "purl:pkg:pypi/urllib3", + "component_name": "urllib3", + "finding_bucket": "new_package", + "suppression_reason": null + } + ], + "suppressed_findings": [], + "rule_catalog": { + "new_package": { + "rule_id": "new_package", + "kind": "risk_finding", + "description": "Component is present only in the after input.", + "finding_buckets": [ + "new_package" + ] + }, + "major_upgrade": { + "rule_id": "major_upgrade", + "kind": "risk_finding", + "description": "Version change is a parseable SemVer major upgrade.", + "finding_buckets": [ + "major_upgrade" + ] + }, + "version_change_unclassified": { + "rule_id": "version_change_unclassified", + "kind": "risk_finding", + "description": "Version changed but could not be classified as a reliable major SemVer upgrade.", + "finding_buckets": [ + "version_change_unclassified" + ] + }, + "unknown_license": { + "rule_id": "unknown_license", + "kind": "risk_finding", + "description": "License metadata is missing, empty, UNKNOWN, or NOASSERTION.", + "finding_buckets": [ + "unknown_license" + ] + }, + "suspicious_source": { + "rule_id": "suspicious_source", + "kind": "risk_finding", + "description": "Source provenance is missing or points to a suspicious scheme, path, or host.", + "finding_buckets": [ + "suspicious_source" + ] + }, + "stale_package": { + "rule_id": "stale_package", + "kind": "risk_finding", + "description": "Staleness check result. Offline mode maps this rule to not_evaluated instead of guessing.", + "finding_buckets": [ + "stale_package", + "not_evaluated" + ] + }, + "max_added_packages": { + "rule_id": "max_added_packages", + "kind": "policy_check", + "description": "Added package count exceeded the configured deterministic threshold.", + "finding_buckets": [] + }, + "allow_sources": { + "rule_id": "allow_sources", + "kind": "policy_check", + "description": "Component source host was not present in the configured allow_sources list.", + "finding_buckets": [] + } + }, + "metadata": { + "before_format": "cyclonedx-json", + "after_format": "cyclonedx-json", + "generated_at": null, + "strict": false, + "stub": false, + "policy_evaluation": { + "applied": true, + "policy_path": "examples\\policy-minimal.yml", + "effective_policy": { + "version": 1, + "block_on": [ + "unknown_license" + ], + "warn_on": [ + "new_package" + ], + "max_added_packages": null, + "allow_sources": [], + "ignore_rules": [] + }, + "blocking_violations": [], + "warning_violations": [ + { + "rule_id": "new_package", + "level": "warn", + "message": "Component was not present in the before input.", + "component_key": "purl:pkg:pypi/urllib3", + "component_name": "urllib3", + "finding_bucket": "new_package", + "suppression_reason": null + } + ], + "suppressed_violations": [], + "totals": { + "blocking": 0, + "warning": 1, + "suppressed": 0, + "ignored_checks": 0 + }, + "exit_code": 0 + } + }, + "notes": [ + "This tool uses heuristic risk classification.", + "No network enrichment was performed." + ] +} diff --git a/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.md b/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.md new file mode 100644 index 0000000..9422e2b --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.md @@ -0,0 +1,62 @@ +# sbom-diff-and-risk report + +## Summary +- Before format: cyclonedx-json +- After format: cyclonedx-json +- Added: 1 +- Removed: 0 +- Version changes: 1 + +## Risk buckets +- new_package: 1 +- major_upgrade: 0 +- version_change_unclassified: 1 +- unknown_license: 0 +- stale_package: 0 +- suspicious_source: 0 +- not_evaluated: 2 + +## Policy summary +- Applied: yes +- Policy path: examples\policy-minimal.yml +- Exit code: 0 +- Blocking findings: 0 +- Warnings: 1 +- Suppressed findings: 0 + +## Added components +| name | version | ecosystem | risk buckets | +|------|---------|-----------|--------------| +| urllib3 | 2.2.1 | pypi | new_package, not_evaluated | + +## Removed components +| name | version | ecosystem | +|------|---------|-----------| +| _none_ | | | + +## Version changes +| name | before | after | classification | risk buckets | +|------|--------|-------|----------------|--------------| +| requests | 2.31.0 | 2.32.0 | version_changed | not_evaluated, version_change_unclassified | + +## Risk findings +| bucket | component | version | rationale | +|--------|-----------|---------|-----------| +| new_package | urllib3 | 2.2.1 | Component was not present in the before input. | +| not_evaluated | requests | 2.32.0 | stale_package was not evaluated because enrichment mode is disabled. | +| not_evaluated | urllib3 | 2.2.1 | stale_package was not evaluated because enrichment mode is disabled. | +| version_change_unclassified | requests | 2.32.0 | Version changed but did not qualify as a parseable SemVer major upgrade. | + +## Blocking violations +| rule id | component | level | message | +|---------|-----------|-------|---------| +| _none_ | | | | + +## Warnings +| rule id | component | level | message | +|---------|-----------|-------|---------| +| new_package | urllib3 | warn | Component was not present in the before input. | + +## Notes +- This tool uses heuristic risk classification. +- No network enrichment was performed. diff --git a/tools/sbom-diff-and-risk/examples/sample-report.json b/tools/sbom-diff-and-risk/examples/sample-report.json index e680946..6c4f73a 100644 --- a/tools/sbom-diff-and-risk/examples/sample-report.json +++ b/tools/sbom-diff-and-risk/examples/sample-report.json @@ -300,6 +300,87 @@ "rationale": "Version changed but did not qualify as a parseable SemVer major upgrade." } ], + "policy_evaluation": { + "applied": false, + "policy_path": null, + "effective_policy": null, + "blocking_violations": [], + "warning_violations": [], + "suppressed_violations": [], + "totals": { + "blocking": 0, + "warning": 0, + "suppressed": 0, + "ignored_checks": 0 + }, + "exit_code": 0 + }, + "blocking_findings": [], + "warning_findings": [], + "suppressed_findings": [], + "rule_catalog": { + "new_package": { + "rule_id": "new_package", + "kind": "risk_finding", + "description": "Component is present only in the after input.", + "finding_buckets": [ + "new_package" + ] + }, + "major_upgrade": { + "rule_id": "major_upgrade", + "kind": "risk_finding", + "description": "Version change is a parseable SemVer major upgrade.", + "finding_buckets": [ + "major_upgrade" + ] + }, + "version_change_unclassified": { + "rule_id": "version_change_unclassified", + "kind": "risk_finding", + "description": "Version changed but could not be classified as a reliable major SemVer upgrade.", + "finding_buckets": [ + "version_change_unclassified" + ] + }, + "unknown_license": { + "rule_id": "unknown_license", + "kind": "risk_finding", + "description": "License metadata is missing, empty, UNKNOWN, or NOASSERTION.", + "finding_buckets": [ + "unknown_license" + ] + }, + "suspicious_source": { + "rule_id": "suspicious_source", + "kind": "risk_finding", + "description": "Source provenance is missing or points to a suspicious scheme, path, or host.", + "finding_buckets": [ + "suspicious_source" + ] + }, + "stale_package": { + "rule_id": "stale_package", + "kind": "risk_finding", + "description": "Staleness check result. Offline mode maps this rule to not_evaluated instead of guessing.", + "finding_buckets": [ + "stale_package", + "not_evaluated" + ] + }, + "max_added_packages": { + "rule_id": "max_added_packages", + "kind": "policy_check", + "description": "Added package count exceeded the configured deterministic threshold.", + "finding_buckets": [] + }, + "allow_sources": { + "rule_id": "allow_sources", + "kind": "policy_check", + "description": "Component source host was not present in the configured allow_sources list.", + "finding_buckets": [] + } + }, "metadata": { "before_format": "cyclonedx-json", "after_format": "cyclonedx-json", @@ -312,9 +393,11 @@ "effective_policy": null, "blocking_violations": [], "warning_violations": [], + "suppressed_violations": [], "totals": { "blocking": 0, "warning": 0, + "suppressed": 0, "ignored_checks": 0 }, "exit_code": 0 diff --git a/tools/sbom-diff-and-risk/examples/sample-report.md b/tools/sbom-diff-and-risk/examples/sample-report.md index c1374ab..6569589 100644 --- a/tools/sbom-diff-and-risk/examples/sample-report.md +++ b/tools/sbom-diff-and-risk/examples/sample-report.md @@ -16,6 +16,14 @@ - suspicious_source: 0 - not_evaluated: 2 +## Policy summary +- Applied: no +- Policy path: none +- Exit code: 0 +- Blocking findings: 0 +- Warnings: 0 +- Suppressed findings: 0 + ## Added components | name | version | ecosystem | risk buckets | |------|---------|-----------|--------------| @@ -39,6 +47,16 @@ | not_evaluated | urllib3 | 2.2.1 | stale_package was not evaluated because enrichment mode is disabled. | | version_change_unclassified | requests | 2.32.0 | Version changed but did not qualify as a parseable SemVer major upgrade. | +## Blocking violations +| rule id | component | level | message | +|---------|-----------|-------|---------| +| _none_ | | | | + +## Warnings +| rule id | component | level | message | +|---------|-----------|-------|---------| +| _none_ | | | | + ## Notes - This tool uses heuristic risk classification. - No network enrichment was performed. diff --git a/tools/sbom-diff-and-risk/examples/sample-requirements-report.json b/tools/sbom-diff-and-risk/examples/sample-requirements-report.json index 6664bfd..4d8427e 100644 --- a/tools/sbom-diff-and-risk/examples/sample-requirements-report.json +++ b/tools/sbom-diff-and-risk/examples/sample-requirements-report.json @@ -236,6 +236,87 @@ "rationale": "Version changed but did not qualify as a parseable SemVer major upgrade." } ], + "policy_evaluation": { + "applied": false, + "policy_path": null, + "effective_policy": null, + "blocking_violations": [], + "warning_violations": [], + "suppressed_violations": [], + "totals": { + "blocking": 0, + "warning": 0, + "suppressed": 0, + "ignored_checks": 0 + }, + "exit_code": 0 + }, + "blocking_findings": [], + "warning_findings": [], + "suppressed_findings": [], + "rule_catalog": { + "new_package": { + "rule_id": "new_package", + "kind": "risk_finding", + "description": "Component is present only in the after input.", + "finding_buckets": [ + "new_package" + ] + }, + "major_upgrade": { + "rule_id": "major_upgrade", + "kind": "risk_finding", + "description": "Version change is a parseable SemVer major upgrade.", + "finding_buckets": [ + "major_upgrade" + ] + }, + "version_change_unclassified": { + "rule_id": "version_change_unclassified", + "kind": "risk_finding", + "description": "Version changed but could not be classified as a reliable major SemVer upgrade.", + "finding_buckets": [ + "version_change_unclassified" + ] + }, + "unknown_license": { + "rule_id": "unknown_license", + "kind": "risk_finding", + "description": "License metadata is missing, empty, UNKNOWN, or NOASSERTION.", + "finding_buckets": [ + "unknown_license" + ] + }, + "suspicious_source": { + "rule_id": "suspicious_source", + "kind": "risk_finding", + "description": "Source provenance is missing or points to a suspicious scheme, path, or host.", + "finding_buckets": [ + "suspicious_source" + ] + }, + "stale_package": { + "rule_id": "stale_package", + "kind": "risk_finding", + "description": "Staleness check result. Offline mode maps this rule to not_evaluated instead of guessing.", + "finding_buckets": [ + "stale_package", + "not_evaluated" + ] + }, + "max_added_packages": { + "rule_id": "max_added_packages", + "kind": "policy_check", + "description": "Added package count exceeded the configured deterministic threshold.", + "finding_buckets": [] + }, + "allow_sources": { + "rule_id": "allow_sources", + "kind": "policy_check", + "description": "Component source host was not present in the configured allow_sources list.", + "finding_buckets": [] + } + }, "metadata": { "before_format": "requirements-txt", "after_format": "requirements-txt", @@ -248,9 +329,11 @@ "effective_policy": null, "blocking_violations": [], "warning_violations": [], + "suppressed_violations": [], "totals": { "blocking": 0, "warning": 0, + "suppressed": 0, "ignored_checks": 0 }, "exit_code": 0 diff --git a/tools/sbom-diff-and-risk/examples/sample-requirements-report.md b/tools/sbom-diff-and-risk/examples/sample-requirements-report.md index cbf2afd..9564d50 100644 --- a/tools/sbom-diff-and-risk/examples/sample-requirements-report.md +++ b/tools/sbom-diff-and-risk/examples/sample-requirements-report.md @@ -16,6 +16,14 @@ - suspicious_source: 0 - not_evaluated: 2 +## Policy summary +- Applied: no +- Policy path: none +- Exit code: 0 +- Blocking findings: 0 +- Warnings: 0 +- Suppressed findings: 0 + ## Added components | name | version | ecosystem | risk buckets | |------|---------|-----------|--------------| @@ -41,6 +49,16 @@ | unknown_license | urllib3 | 2.2.1 | License is missing, empty, UNKNOWN, or NOASSERTION. | | version_change_unclassified | requests | 2.32.0 | Version changed but did not qualify as a parseable SemVer major upgrade. | +## Blocking violations +| rule id | component | level | message | +|---------|-----------|-------|---------| +| _none_ | | | | + +## Warnings +| rule id | component | level | message | +|---------|-----------|-------|---------| +| _none_ | | | | + ## Notes - This tool uses heuristic risk classification. - No network enrichment was performed. diff --git a/tools/sbom-diff-and-risk/examples/sample-sarif.sarif b/tools/sbom-diff-and-risk/examples/sample-sarif.sarif new file mode 100644 index 0000000..f23d889 --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/sample-sarif.sarif @@ -0,0 +1,300 @@ +{ + "$schema": "https://json.schemastore.org/sarif-2.1.0.json", + "version": "2.1.0", + "runs": [ + { + "tool": { + "driver": { + "name": "sbom-diff-risk", + "fullName": "sbom-diff-risk", + "version": "0.1.0", + "semanticVersion": "0.1.0", + "rules": [ + { + "id": "sdr.major_upgrade", + "name": "major_upgrade", + "shortDescription": { + "text": "Version change is a parseable SemVer major upgrade." + }, + "fullDescription": { + "text": "Version change is a parseable SemVer major upgrade." + }, + "defaultConfiguration": { + "level": "note" + }, + "properties": { + "tags": [ + "supply-chain", + "sbom" + ] + } + }, + { + "id": "sdr.policy_violation.allow_sources", + "name": "policy_violation.allow_sources", + "shortDescription": { + "text": "Blocking policy violation: allow_sources" + }, + "fullDescription": { + "text": "Component source host was not present in the configured allow_sources list." + }, + "defaultConfiguration": { + "level": "error" + }, + "properties": { + "tags": [ + "supply-chain", + "policy" + ] + } + }, + { + "id": "sdr.policy_violation.max_added_packages", + "name": "policy_violation.max_added_packages", + "shortDescription": { + "text": "Blocking policy violation: max_added_packages" + }, + "fullDescription": { + "text": "Added package count exceeded the configured deterministic threshold." + }, + "defaultConfiguration": { + "level": "error" + }, + "properties": { + "tags": [ + "supply-chain", + "policy" + ] + } + }, + { + "id": "sdr.suspicious_source", + "name": "suspicious_source", + "shortDescription": { + "text": "Source provenance is missing or points to a suspicious scheme, path, or host." + }, + "fullDescription": { + "text": "Source provenance is missing or points to a suspicious scheme, path, or host." + }, + "defaultConfiguration": { + "level": "warning" + }, + "properties": { + "tags": [ + "supply-chain", + "sbom" + ] + } + }, + { + "id": "sdr.unknown_license", + "name": "unknown_license", + "shortDescription": { + "text": "License metadata is missing, empty, UNKNOWN, or NOASSERTION." + }, + "fullDescription": { + "text": "License metadata is missing, empty, UNKNOWN, or NOASSERTION." + }, + "defaultConfiguration": { + "level": "warning" + }, + "properties": { + "tags": [ + "supply-chain", + "sbom" + ] + } + } + ] + } + }, + "artifacts": [ + { + "location": { + "uri": "examples/sarif_before.json", + "uriBaseId": "%SRCROOT%" + } + }, + { + "location": { + "uri": "examples/sarif_after.json", + "uriBaseId": "%SRCROOT%" + } + } + ], + "properties": { + "sbom_diff_risk": { + "result_limit": 5000, + "total_candidate_results": 5, + "emitted_results": 5, + "omitted_results": 0, + "truncated": false, + "prioritization": "error results first, then warning, then note; direct mapped findings before policy-only checks; stable rule priority and component key tie-breakers.", + "warning": null + } + }, + "results": [ + { + "ruleId": "sdr.suspicious_source", + "level": "error", + "message": { + "text": "Blocked by policy: mystery-lib 0.1.0 has suspicious or incomplete source provenance." + }, + "locations": [ + { + "physicalLocation": { + "artifactLocation": { + "uri": "examples/sarif_after.json", + "uriBaseId": "%SRCROOT%" + }, + "region": { + "startLine": 1 + } + } + } + ], + "partialFingerprints": { + "ruleId": "sdr.suspicious_source", + "componentKey": "purl:pkg:pypi/mystery-lib" + }, + "properties": { + "component_key": "purl:pkg:pypi/mystery-lib", + "component_name": "mystery-lib", + "finding_bucket": "suspicious_source", + "policy_blocking": true, + "result_kind": "risk_finding", + "blocking_rule_id": "suspicious_source" + } + }, + { + "ruleId": "sdr.unknown_license", + "level": "error", + "message": { + "text": "Blocked by policy: mystery-lib 0.1.0 has missing or unknown license metadata." + }, + "locations": [ + { + "physicalLocation": { + "artifactLocation": { + "uri": "examples/sarif_after.json", + "uriBaseId": "%SRCROOT%" + }, + "region": { + "startLine": 1 + } + } + } + ], + "partialFingerprints": { + "ruleId": "sdr.unknown_license", + "componentKey": "purl:pkg:pypi/mystery-lib" + }, + "properties": { + "component_key": "purl:pkg:pypi/mystery-lib", + "component_name": "mystery-lib", + "finding_bucket": "unknown_license", + "policy_blocking": true, + "result_kind": "risk_finding", + "blocking_rule_id": "unknown_license" + } + }, + { + "ruleId": "sdr.policy_violation.allow_sources", + "level": "error", + "message": { + "text": "mystery-lib: Source host 198.51.100.10 is not present in allow_sources." + }, + "locations": [ + { + "physicalLocation": { + "artifactLocation": { + "uri": "examples/sarif_after.json", + "uriBaseId": "%SRCROOT%" + }, + "region": { + "startLine": 1 + } + } + } + ], + "partialFingerprints": { + "ruleId": "sdr.policy_violation.allow_sources", + "componentKey": "purl:pkg:pypi/mystery-lib" + }, + "properties": { + "policy_rule_id": "allow_sources", + "component_key": "purl:pkg:pypi/mystery-lib", + "component_name": "mystery-lib", + "result_kind": "policy_violation" + } + }, + { + "ruleId": "sdr.policy_violation.max_added_packages", + "level": "error", + "message": { + "text": "Added package count 1 exceeds max_added_packages=0." + }, + "locations": [ + { + "physicalLocation": { + "artifactLocation": { + "uri": "examples/sarif_after.json", + "uriBaseId": "%SRCROOT%" + }, + "region": { + "startLine": 1 + } + } + } + ], + "partialFingerprints": { + "ruleId": "sdr.policy_violation.max_added_packages", + "componentKey": "global-policy-check" + }, + "properties": { + "policy_rule_id": "max_added_packages", + "component_key": null, + "component_name": null, + "result_kind": "policy_violation" + } + }, + { + "ruleId": "sdr.major_upgrade", + "level": "note", + "message": { + "text": "Version changed from 1.9.0 to 2.0.0 with a higher major version." + }, + "locations": [ + { + "physicalLocation": { + "artifactLocation": { + "uri": "examples/sarif_after.json", + "uriBaseId": "%SRCROOT%" + }, + "region": { + "startLine": 1 + } + } + } + ], + "partialFingerprints": { + "ruleId": "sdr.major_upgrade", + "componentKey": "purl:pkg:pypi/requests" + }, + "properties": { + "component_key": "purl:pkg:pypi/requests", + "component_name": "requests", + "finding_bucket": "major_upgrade", + "policy_blocking": false, + "result_kind": "risk_finding" + } + } + ], + "originalUriBaseIds": { + "%SRCROOT%": { + "uri": "file:///D:/OneDrive/Code/scientific-computing-toolkit-real/tools/sbom-diff-and-risk/" + } + } + } + ] +} diff --git a/tools/sbom-diff-and-risk/examples/sarif_after.json b/tools/sbom-diff-and-risk/examples/sarif_after.json new file mode 100644 index 0000000..b4ce02c --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/sarif_after.json @@ -0,0 +1,40 @@ +{ + "bomFormat": "CycloneDX", + "specVersion": "1.5", + "version": 1, + "components": [ + { + "bom-ref": "pkg:pypi/requests@2.0.0", + "type": "library", + "name": "requests", + "version": "2.0.0", + "purl": "pkg:pypi/requests@2.0.0", + "licenses": [ + { + "license": { + "id": "Apache-2.0" + } + } + ], + "externalReferences": [ + { + "type": "website", + "url": "https://pypi.org/project/requests/" + } + ] + }, + { + "bom-ref": "pkg:pypi/mystery-lib@0.1.0", + "type": "library", + "name": "mystery-lib", + "version": "0.1.0", + "purl": "pkg:pypi/mystery-lib@0.1.0", + "externalReferences": [ + { + "type": "distribution", + "url": "http://198.51.100.10/packages/mystery-lib-0.1.0.tar.gz" + } + ] + } + ] +} diff --git a/tools/sbom-diff-and-risk/examples/sarif_before.json b/tools/sbom-diff-and-risk/examples/sarif_before.json new file mode 100644 index 0000000..060b258 --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/sarif_before.json @@ -0,0 +1,27 @@ +{ + "bomFormat": "CycloneDX", + "specVersion": "1.5", + "version": 1, + "components": [ + { + "bom-ref": "pkg:pypi/requests@1.9.0", + "type": "library", + "name": "requests", + "version": "1.9.0", + "purl": "pkg:pypi/requests@1.9.0", + "licenses": [ + { + "license": { + "id": "Apache-2.0" + } + } + ], + "externalReferences": [ + { + "type": "website", + "url": "https://pypi.org/project/requests/" + } + ] + } + ] +} diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/cli.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/cli.py index 64714b9..41aa8ec 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/cli.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/cli.py @@ -6,12 +6,15 @@ from typing import Sequence from .diffing import diff_components +from .errors import ParseError, PolicyError from .models import CompareReport, ReportComponents, ReportMetadata, ReportSummary from .normalize import SUPPORTED_FORMATS, normalize_input from .policy_evaluator import evaluate_policy from .policy_parser import build_policy +from .presentation import effective_policy_evaluation, summarize_violations_by_rule from .report_json import render_report_json from .report_md import render_report_markdown +from .report_sarif import render_report_sarif_output from .risk import evaluate_risks, summarize_risks @@ -25,6 +28,8 @@ def build_parser() -> argparse.ArgumentParser: compare = subparsers.add_parser( "compare", help="Compare two dependency inputs and write JSON and/or Markdown reports.", + description="Compare two local dependency inputs and emit deterministic reports.", + epilog="Exit codes: 0 = success/no blocking violations, 1 = blocking policy violations, 2 = usage/parse/runtime error.", ) compare.add_argument("--before", type=Path, required=True, help="Path to the before input.") compare.add_argument("--after", type=Path, required=True, help="Path to the after input.") @@ -48,10 +53,33 @@ def build_parser() -> argparse.ArgumentParser: ) compare.add_argument("--out-json", type=Path, default=None, help="Write a JSON report to this path.") compare.add_argument("--out-md", type=Path, default=None, help="Write a Markdown report to this path.") - compare.add_argument("--policy", type=Path, default=None, help="Path to a YAML policy file.") - compare.add_argument("--fail-on", default=None, help="Comma-separated policy rule ids that should block.") - compare.add_argument("--warn-on", default=None, help="Comma-separated policy rule ids that should warn.") - compare.add_argument("--strict", action="store_true", help="Treat scaffold notes as errors.") + compare.add_argument( + "--out-sarif", + type=Path, + default=None, + help="Write a SARIF 2.1.0 subset report for GitHub-compatible code scanning ingestion.", + ) + compare.add_argument( + "--policy", + type=Path, + default=None, + help="Apply a YAML policy v1 file. Blocking violations return exit code 1.", + ) + compare.add_argument( + "--fail-on", + default=None, + help="Comma-separated rule ids to treat as blocking, merged with any --policy block_on values.", + ) + compare.add_argument( + "--warn-on", + default=None, + help="Comma-separated rule ids to treat as warnings, merged with any --policy warn_on values.", + ) + compare.add_argument( + "--strict", + action="store_true", + help="Fail with exit code 2 if normalization emits warnings or conservative parser notes.", + ) compare.add_argument( "--enrich-pypi", action="store_true", @@ -71,9 +99,10 @@ def main(argv: Sequence[str] | None = None) -> int: args = parser.parse_args(argv) try: return args.handler(args) - except (FileNotFoundError, NotImplementedError, ValueError) as exc: + except (FileNotFoundError, NotImplementedError, ParseError, PolicyError, ValueError) as exc: parser.print_usage(sys.stderr) print(f"{parser.prog}: error: {exc}", file=sys.stderr) + print(f"{parser.prog}: command failed with exit code 2 before report completion.", file=sys.stderr) return 2 @@ -81,8 +110,8 @@ def run_compare(args: argparse.Namespace) -> int: if args.enrich_pypi: raise NotImplementedError("--enrich-pypi is reserved for a later network-enabled release.") - if args.out_json is None and args.out_md is None: - raise ValueError("at least one of --out-json or --out-md must be provided") + if args.out_json is None and args.out_md is None and args.out_sarif is None: + raise ValueError("at least one of --out-json, --out-md, or --out-sarif must be provided") before_path: Path = args.before after_path: Path = args.after @@ -136,12 +165,25 @@ def run_compare(args: argparse.Namespace) -> int: ) if args.strict and (before_notes or after_notes): - raise ValueError("strict mode failed because normalization produced warnings.") + raise ValueError(_format_strict_failure(before_notes, after_notes)) if args.out_json is not None: _write_text(args.out_json, render_report_json(report)) if args.out_md is not None: _write_text(args.out_md, render_report_markdown(report)) + if args.out_sarif is not None: + sarif_output = render_report_sarif_output( + report, + before_path=before_path, + after_path=after_path, + base_dir=Path.cwd(), + ) + _write_text(args.out_sarif, sarif_output.content) + if sarif_output.metadata.warning_message: + print(f"sbom-diff-risk: warning: {sarif_output.metadata.warning_message}", file=sys.stderr) + + if policy_evaluation.exit_code == 1: + print(_format_policy_failure_summary(policy_evaluation), file=sys.stderr) return policy_evaluation.exit_code @@ -159,5 +201,33 @@ def _write_text(path: Path, content: str) -> None: path.write_text(content, encoding="utf-8") +def _format_strict_failure(before_notes: list[str], after_notes: list[str]) -> str: + notes = [*before_notes, *after_notes] + if not notes: + return "strict mode failed because normalization produced warnings." + return f"strict mode failed because normalization produced {len(notes)} warning(s): {notes[0]}" + + +def _format_policy_failure_summary(policy_evaluation) -> str: + resolved = effective_policy_evaluation(policy_evaluation) + lines = [ + "sbom-diff-risk: blocking policy violations detected (exit code 1).", + ( + f"sbom-diff-risk: policy source = {resolved.policy_path}" + if resolved.policy_path + else "sbom-diff-risk: policy source = CLI flags" + ), + f"sbom-diff-risk: blocking findings = {len(resolved.blocking_violations)}", + ] + for rule_id, count in summarize_violations_by_rule(resolved.blocking_violations): + lines.append(f"sbom-diff-risk: {rule_id} = {count}") + if resolved.warning_violations: + lines.append(f"sbom-diff-risk: warnings = {len(resolved.warning_violations)}") + if resolved.suppressed_violations: + lines.append(f"sbom-diff-risk: suppressed = {len(resolved.suppressed_violations)}") + lines.append("sbom-diff-risk: outputs were written; inspect the generated reports for full details.") + return "\n".join(lines) + + if __name__ == "__main__": raise SystemExit(main()) diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_evaluator.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_evaluator.py index 86050ce..c52ff0d 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_evaluator.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_evaluator.py @@ -20,21 +20,33 @@ def evaluate_policy( blocking_violations: list[PolicyViolation] = [] warning_violations: list[PolicyViolation] = [] + suppressed_violations: list[PolicyViolation] = [] ignored_checks = 0 for finding in findings: rule_id = finding_rule_id(finding) + severity = _severity_for_rule(policy, rule_id) if rule_id in policy.ignore_rules: ignored_checks += 1 + suppressed_violations.append( + PolicyViolation( + rule_id=rule_id, + level=severity, + message=finding.rationale, + component_key=finding.component_key, + component_name=finding.component.name, + finding_bucket=finding.bucket.value, + suppression_reason="ignored_by_policy", + ) + ) continue - level = _severity_for_rule(policy, rule_id) - if level is None: + if severity is None: continue violation = PolicyViolation( rule_id=rule_id, - level=level, + level=severity, message=finding.rationale, component_key=finding.component_key, component_name=finding.component.name, @@ -44,20 +56,18 @@ def evaluate_policy( if policy.max_added_packages is not None and len(added) > policy.max_added_packages: rule_id = "max_added_packages" + severity = _severity_for_rule(policy, rule_id, default=PolicyLevel.BLOCK) + violation = PolicyViolation( + rule_id=rule_id, + level=severity, + message=f"Added package count {len(added)} exceeds max_added_packages={policy.max_added_packages}.", + suppression_reason="ignored_by_policy" if rule_id in policy.ignore_rules else None, + ) if rule_id in policy.ignore_rules: ignored_checks += 1 - else: - level = _severity_for_rule(policy, rule_id, default=PolicyLevel.BLOCK) - if level is not None: - violation = PolicyViolation( - rule_id=rule_id, - level=level, - message=( - f"Added package count {len(added)} exceeds max_added_packages=" - f"{policy.max_added_packages}." - ), - ) - _append_violation(violation, blocking_violations, warning_violations) + suppressed_violations.append(violation) + elif severity is not None: + _append_violation(violation, blocking_violations, warning_violations) if policy.allow_sources: for component in _components_for_source_policy(added, changed): @@ -66,22 +76,25 @@ def evaluate_policy( continue rule_id = "allow_sources" - if rule_id in policy.ignore_rules: - ignored_checks += 1 - continue - - level = _severity_for_rule(policy, rule_id, default=PolicyLevel.BLOCK) - if level is None: - continue - + severity = _severity_for_rule(policy, rule_id, default=PolicyLevel.BLOCK) violation = PolicyViolation( rule_id=rule_id, - level=level, - message=f"Source host {host} is not present in allow_sources.", + level=severity, + message=f"Source host {host or 'missing'} is not present in allow_sources.", component_key=component_key(component), component_name=component.name, + suppression_reason="ignored_by_policy" if rule_id in policy.ignore_rules else None, ) - _append_violation(violation, blocking_violations, warning_violations) + if rule_id in policy.ignore_rules: + ignored_checks += 1 + suppressed_violations.append(violation) + continue + if severity is not None: + _append_violation(violation, blocking_violations, warning_violations) + + blocking_violations.sort(key=_violation_sort_key) + warning_violations.sort(key=_violation_sort_key) + suppressed_violations.sort(key=_violation_sort_key) exit_code = 1 if blocking_violations else 0 return PolicyEvaluation( @@ -90,6 +103,7 @@ def evaluate_policy( effective_policy=policy, blocking_violations=blocking_violations, warning_violations=warning_violations, + suppressed_violations=suppressed_violations, ignored_checks=ignored_checks, exit_code=exit_code, ) @@ -121,7 +135,7 @@ def _append_violation( ) -> None: if violation.level is PolicyLevel.BLOCK: blocking_violations.append(violation) - else: + elif violation.level is PolicyLevel.WARN: warning_violations.append(violation) @@ -136,3 +150,11 @@ def _source_host(source_url: str | None) -> str | None: return None host = (urlparse(source_url).hostname or "").strip().lower() return host or None + + +def _violation_sort_key(violation: PolicyViolation) -> tuple[str, str, str]: + return ( + violation.rule_id, + violation.component_key or "", + violation.component_name or "", + ) diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_models.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_models.py index 20f7d4c..ad2362f 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_models.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_models.py @@ -34,11 +34,12 @@ class PolicyConfig: @dataclass(slots=True) class PolicyViolation: rule_id: str - level: PolicyLevel + level: PolicyLevel | None message: str component_key: str | None = None component_name: str | None = None finding_bucket: str | None = None + suppression_reason: str | None = None @dataclass(slots=True) @@ -48,5 +49,6 @@ class PolicyEvaluation: effective_policy: PolicyConfig | None = None blocking_violations: list[PolicyViolation] = field(default_factory=list) warning_violations: list[PolicyViolation] = field(default_factory=list) + suppressed_violations: list[PolicyViolation] = field(default_factory=list) ignored_checks: int = 0 exit_code: int = 0 diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/presentation.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/presentation.py new file mode 100644 index 0000000..421f661 --- /dev/null +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/presentation.py @@ -0,0 +1,150 @@ +from __future__ import annotations + +from dataclasses import dataclass +from typing import Any + +from .policy_models import PolicyConfig, PolicyEvaluation, PolicyViolation + + +@dataclass(slots=True, frozen=True) +class RuleCatalogEntry: + rule_id: str + kind: str + description: str + finding_buckets: tuple[str, ...] = () + + +_RULE_CATALOG = ( + RuleCatalogEntry( + rule_id="new_package", + kind="risk_finding", + description="Component is present only in the after input.", + finding_buckets=("new_package",), + ), + RuleCatalogEntry( + rule_id="major_upgrade", + kind="risk_finding", + description="Version change is a parseable SemVer major upgrade.", + finding_buckets=("major_upgrade",), + ), + RuleCatalogEntry( + rule_id="version_change_unclassified", + kind="risk_finding", + description="Version changed but could not be classified as a reliable major SemVer upgrade.", + finding_buckets=("version_change_unclassified",), + ), + RuleCatalogEntry( + rule_id="unknown_license", + kind="risk_finding", + description="License metadata is missing, empty, UNKNOWN, or NOASSERTION.", + finding_buckets=("unknown_license",), + ), + RuleCatalogEntry( + rule_id="suspicious_source", + kind="risk_finding", + description="Source provenance is missing or points to a suspicious scheme, path, or host.", + finding_buckets=("suspicious_source",), + ), + RuleCatalogEntry( + rule_id="stale_package", + kind="risk_finding", + description="Staleness check result. Offline mode maps this rule to not_evaluated instead of guessing.", + finding_buckets=("stale_package", "not_evaluated"), + ), + RuleCatalogEntry( + rule_id="max_added_packages", + kind="policy_check", + description="Added package count exceeded the configured deterministic threshold.", + ), + RuleCatalogEntry( + rule_id="allow_sources", + kind="policy_check", + description="Component source host was not present in the configured allow_sources list.", + ), +) + + +def build_policy_report_sections(policy_evaluation: PolicyEvaluation | None) -> dict[str, Any]: + evaluation_dict = policy_evaluation_to_dict(policy_evaluation) + return { + "policy_evaluation": evaluation_dict, + "blocking_findings": [ + policy_violation_to_dict(item) for item in effective_policy_evaluation(policy_evaluation).blocking_violations + ], + "warning_findings": [ + policy_violation_to_dict(item) for item in effective_policy_evaluation(policy_evaluation).warning_violations + ], + "suppressed_findings": [ + policy_violation_to_dict(item) for item in effective_policy_evaluation(policy_evaluation).suppressed_violations + ], + "rule_catalog": rule_catalog_to_dict(), + } + + +def effective_policy_evaluation(policy_evaluation: PolicyEvaluation | None) -> PolicyEvaluation: + if policy_evaluation is not None: + return policy_evaluation + return PolicyEvaluation(applied=False, policy_path=None, effective_policy=None, exit_code=0) + + +def policy_evaluation_to_dict(policy_evaluation: PolicyEvaluation | None) -> dict[str, Any]: + resolved = effective_policy_evaluation(policy_evaluation) + return { + "applied": resolved.applied, + "policy_path": resolved.policy_path, + "effective_policy": policy_config_to_dict(resolved.effective_policy), + "blocking_violations": [policy_violation_to_dict(item) for item in resolved.blocking_violations], + "warning_violations": [policy_violation_to_dict(item) for item in resolved.warning_violations], + "suppressed_violations": [policy_violation_to_dict(item) for item in resolved.suppressed_violations], + "totals": { + "blocking": len(resolved.blocking_violations), + "warning": len(resolved.warning_violations), + "suppressed": len(resolved.suppressed_violations), + "ignored_checks": resolved.ignored_checks, + }, + "exit_code": resolved.exit_code, + } + + +def policy_config_to_dict(policy: PolicyConfig | None) -> dict[str, Any] | None: + if policy is None: + return None + return { + "version": policy.version, + "block_on": list(policy.block_on), + "warn_on": list(policy.warn_on), + "max_added_packages": policy.max_added_packages, + "allow_sources": list(policy.allow_sources), + "ignore_rules": list(policy.ignore_rules), + } + + +def policy_violation_to_dict(violation: PolicyViolation) -> dict[str, Any]: + return { + "rule_id": violation.rule_id, + "level": violation.level.value if violation.level is not None else None, + "message": violation.message, + "component_key": violation.component_key, + "component_name": violation.component_name, + "finding_bucket": violation.finding_bucket, + "suppression_reason": violation.suppression_reason, + } + + +def rule_catalog_to_dict() -> dict[str, dict[str, Any]]: + return { + entry.rule_id: { + "rule_id": entry.rule_id, + "kind": entry.kind, + "description": entry.description, + "finding_buckets": list(entry.finding_buckets), + } + for entry in _RULE_CATALOG + } + + +def summarize_violations_by_rule(violations: list[PolicyViolation]) -> list[tuple[str, int]]: + counts: dict[str, int] = {} + for violation in violations: + counts[violation.rule_id] = counts.get(violation.rule_id, 0) + 1 + return sorted(counts.items()) diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_json.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_json.py index 99ed3b9..2026dae 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_json.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_json.py @@ -3,10 +3,11 @@ import json from .models import CompareReport, Component, ComponentChange, RiskFinding -from .policy_models import PolicyConfig, PolicyEvaluation, PolicyViolation +from .presentation import build_policy_report_sections def render_report_json(report: CompareReport) -> str: + policy_sections = build_policy_report_sections(report.metadata.policy_evaluation) payload = { "summary": { "added": report.summary.added, @@ -20,13 +21,18 @@ def render_report_json(report: CompareReport) -> str: "changed": [_change_to_dict(change) for change in report.components.changed], }, "risks": [_risk_to_dict(finding) for finding in report.risks], + "policy_evaluation": policy_sections["policy_evaluation"], + "blocking_findings": policy_sections["blocking_findings"], + "warning_findings": policy_sections["warning_findings"], + "suppressed_findings": policy_sections["suppressed_findings"], + "rule_catalog": policy_sections["rule_catalog"], "metadata": { "before_format": report.metadata.before_format, "after_format": report.metadata.after_format, "generated_at": report.metadata.generated_at, "strict": report.metadata.strict, "stub": report.metadata.stub, - "policy_evaluation": _policy_evaluation_to_dict(report.metadata.policy_evaluation), + "policy_evaluation": policy_sections["policy_evaluation"], }, "notes": list(report.notes), } @@ -64,58 +70,3 @@ def _risk_to_dict(finding: RiskFinding) -> dict[str, object]: "component": _component_to_dict(finding.component), "rationale": finding.rationale, } - - -def _policy_evaluation_to_dict(policy_evaluation: PolicyEvaluation | None) -> dict[str, object]: - if policy_evaluation is None: - return { - "applied": False, - "policy_path": None, - "effective_policy": None, - "blocking_violations": [], - "warning_violations": [], - "totals": { - "blocking": 0, - "warning": 0, - "ignored_checks": 0, - }, - "exit_code": 0, - } - - return { - "applied": policy_evaluation.applied, - "policy_path": policy_evaluation.policy_path, - "effective_policy": _policy_config_to_dict(policy_evaluation.effective_policy), - "blocking_violations": [_policy_violation_to_dict(item) for item in policy_evaluation.blocking_violations], - "warning_violations": [_policy_violation_to_dict(item) for item in policy_evaluation.warning_violations], - "totals": { - "blocking": len(policy_evaluation.blocking_violations), - "warning": len(policy_evaluation.warning_violations), - "ignored_checks": policy_evaluation.ignored_checks, - }, - "exit_code": policy_evaluation.exit_code, - } - - -def _policy_config_to_dict(policy: PolicyConfig | None) -> dict[str, object] | None: - if policy is None: - return None - return { - "version": policy.version, - "block_on": list(policy.block_on), - "warn_on": list(policy.warn_on), - "max_added_packages": policy.max_added_packages, - "allow_sources": list(policy.allow_sources), - "ignore_rules": list(policy.ignore_rules), - } - - -def _policy_violation_to_dict(violation: PolicyViolation) -> dict[str, object]: - return { - "rule_id": violation.rule_id, - "level": violation.level.value, - "message": violation.message, - "component_key": violation.component_key, - "component_name": violation.component_name, - "finding_bucket": violation.finding_bucket, - } diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_md.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_md.py index fa2ac49..06416df 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_md.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_md.py @@ -2,9 +2,11 @@ from .diffing import component_key from .models import CompareReport +from .presentation import effective_policy_evaluation def render_report_markdown(report: CompareReport) -> str: + policy_evaluation = effective_policy_evaluation(report.metadata.policy_evaluation) lines = [ "# sbom-diff-and-risk report", "", @@ -21,6 +23,19 @@ def render_report_markdown(report: CompareReport) -> str: for bucket, count in report.summary.risk_counts.items(): lines.append(f"- {bucket}: {count}") + lines.extend( + [ + "", + "## Policy summary", + f"- Applied: {'yes' if policy_evaluation.applied else 'no'}", + f"- Policy path: {policy_evaluation.policy_path or 'none'}", + f"- Exit code: {policy_evaluation.exit_code}", + f"- Blocking findings: {len(policy_evaluation.blocking_violations)}", + f"- Warnings: {len(policy_evaluation.warning_violations)}", + f"- Suppressed findings: {len(policy_evaluation.suppressed_violations)}", + ] + ) + lines.extend( [ "", @@ -87,6 +102,56 @@ def render_report_markdown(report: CompareReport) -> str: else: lines.append("| _none_ | | | |") + lines.extend( + [ + "", + "## Blocking violations", + "| rule id | component | level | message |", + "|---------|-----------|-------|---------|", + ] + ) + if policy_evaluation.blocking_violations: + for violation in policy_evaluation.blocking_violations: + lines.append( + f"| {violation.rule_id} | {violation.component_name or ''} | {violation.level.value if violation.level else ''} | " + f"{_escape_table_text(violation.message)} |" + ) + else: + lines.append("| _none_ | | | |") + + lines.extend( + [ + "", + "## Warnings", + "| rule id | component | level | message |", + "|---------|-----------|-------|---------|", + ] + ) + if policy_evaluation.warning_violations: + for violation in policy_evaluation.warning_violations: + lines.append( + f"| {violation.rule_id} | {violation.component_name or ''} | {violation.level.value if violation.level else ''} | " + f"{_escape_table_text(violation.message)} |" + ) + else: + lines.append("| _none_ | | | |") + + if policy_evaluation.suppressed_violations: + lines.extend( + [ + "", + "## Suppressions", + "| rule id | component | level | reason | message |", + "|---------|-----------|-------|--------|---------|", + ] + ) + for violation in policy_evaluation.suppressed_violations: + lines.append( + f"| {violation.rule_id} | {violation.component_name or ''} | " + f"{violation.level.value if violation.level else 'n/a'} | " + f"{violation.suppression_reason or ''} | {_escape_table_text(violation.message)} |" + ) + lines.extend(["", "## Notes"]) if report.notes: lines.extend(f"- {note}" for note in report.notes) diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_sarif.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_sarif.py new file mode 100644 index 0000000..14bcb19 --- /dev/null +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/report_sarif.py @@ -0,0 +1,417 @@ +from __future__ import annotations + +import json +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +from . import __version__ +from .models import CompareReport, RiskBucket, RiskFinding +from .policy_models import PolicyViolation +from .presentation import effective_policy_evaluation, rule_catalog_to_dict + +DEFAULT_SARIF_RESULT_LIMIT = 5000 +SARIF_PRIORITIZATION_DESCRIPTION = ( + "error results first, then warning, then note; direct mapped findings before policy-only checks; " + "stable rule priority and component key tie-breakers." +) + +_SARIF_SUPPORTED_RISK_BUCKETS = { + RiskBucket.SUSPICIOUS_SOURCE, + RiskBucket.UNKNOWN_LICENSE, + RiskBucket.MAJOR_UPGRADE, +} +_SARIF_POLICY_ONLY_RULE_IDS = {"allow_sources", "max_added_packages"} +_LEVEL_PRIORITY = {"error": 0, "warning": 1, "note": 2} +_RULE_PRIORITY = { + "sdr.suspicious_source": 0, + "sdr.unknown_license": 1, + "sdr.major_upgrade": 2, + "sdr.policy_violation.allow_sources": 3, + "sdr.policy_violation.max_added_packages": 4, +} + + +@dataclass(slots=True, frozen=True) +class SarifRenderMetadata: + result_limit: int + total_candidate_results: int + emitted_results: int + omitted_results: int + truncated: bool + prioritization: str = SARIF_PRIORITIZATION_DESCRIPTION + + @property + def warning_message(self) -> str | None: + if not self.truncated: + return None + return ( + "SARIF results were truncated deterministically for GitHub-oriented compatibility: " + f"emitted {self.emitted_results} of {self.total_candidate_results} candidate results " + f"(limit {self.result_limit})." + ) + + +@dataclass(slots=True, frozen=True) +class SarifRenderOutput: + content: str + metadata: SarifRenderMetadata + + +def render_report_sarif( + report: CompareReport, + *, + before_path: Path, + after_path: Path, + base_dir: Path | None = None, + result_limit: int | None = None, +) -> str: + return render_report_sarif_output( + report, + before_path=before_path, + after_path=after_path, + base_dir=base_dir, + result_limit=result_limit, + ).content + + +def render_report_sarif_output( + report: CompareReport, + *, + before_path: Path, + after_path: Path, + base_dir: Path | None = None, + result_limit: int | None = None, +) -> SarifRenderOutput: + if result_limit is None: + result_limit = DEFAULT_SARIF_RESULT_LIMIT + if result_limit <= 0: + raise ValueError("result_limit must be a positive integer.") + + resolved_base_dir = base_dir.resolve() if base_dir is not None else None + policy_evaluation = effective_policy_evaluation(report.metadata.policy_evaluation) + blocking_map = _blocking_violation_map(policy_evaluation.blocking_violations) + emitted_blocking_keys: set[tuple[str, str | None]] = set() + + candidate_results: list[dict[str, Any]] = [] + + for finding in report.risks: + if finding.bucket not in _SARIF_SUPPORTED_RISK_BUCKETS: + continue + + policy_rule_id = _policy_rule_id_for_bucket(finding.bucket) + blocking_violation = blocking_map.get((policy_rule_id, finding.component_key)) + result = _risk_finding_to_result( + finding, + after_path=after_path, + base_dir=resolved_base_dir, + blocking_violation=blocking_violation, + ) + candidate_results.append(result) + if blocking_violation is not None: + emitted_blocking_keys.add((policy_rule_id, finding.component_key)) + + for violation in policy_evaluation.blocking_violations: + lookup_key = (violation.rule_id, violation.component_key) + if lookup_key in emitted_blocking_keys: + continue + + sarif_rule_id = sarif_rule_id_for_policy_violation(violation.rule_id) + if sarif_rule_id is None: + continue + + result = _policy_violation_to_result( + violation, + after_path=after_path, + base_dir=resolved_base_dir, + ) + candidate_results.append(result) + + candidate_results.sort(key=_result_sort_key) + results = candidate_results[:result_limit] + metadata = SarifRenderMetadata( + result_limit=result_limit, + total_candidate_results=len(candidate_results), + emitted_results=len(results), + omitted_results=max(0, len(candidate_results) - len(results)), + truncated=len(candidate_results) > result_limit, + ) + used_rule_ids = {result["ruleId"] for result in results} + + rules = [_sarif_rule_metadata(rule_id) for rule_id in sorted(used_rule_ids)] + sarif_document: dict[str, Any] = { + "$schema": "https://json.schemastore.org/sarif-2.1.0.json", + "version": "2.1.0", + "runs": [ + { + "tool": { + "driver": { + "name": "sbom-diff-risk", + "fullName": "sbom-diff-risk", + "version": __version__, + "semanticVersion": __version__, + "rules": rules, + } + }, + "artifacts": [ + { + "location": _artifact_location(before_path, resolved_base_dir), + }, + { + "location": _artifact_location(after_path, resolved_base_dir), + }, + ], + "properties": { + "sbom_diff_risk": _guardrail_metadata_to_dict(metadata), + }, + "results": results, + } + ], + } + + if resolved_base_dir is not None: + sarif_document["runs"][0]["originalUriBaseIds"] = { + "%SRCROOT%": { + "uri": _directory_uri(resolved_base_dir), + } + } + + return SarifRenderOutput( + content=json.dumps(sarif_document, indent=2) + "\n", + metadata=metadata, + ) + + +def sarif_rule_id_for_risk_bucket(bucket: RiskBucket) -> str | None: + if bucket not in _SARIF_SUPPORTED_RISK_BUCKETS: + return None + return f"sdr.{bucket.value}" + + +def sarif_rule_id_for_policy_violation(rule_id: str) -> str | None: + if rule_id not in _SARIF_POLICY_ONLY_RULE_IDS: + return None + return f"sdr.policy_violation.{rule_id}" + + +def _risk_finding_to_result( + finding: RiskFinding, + *, + after_path: Path, + base_dir: Path | None, + blocking_violation: PolicyViolation | None, +) -> dict[str, Any]: + rule_id = sarif_rule_id_for_risk_bucket(finding.bucket) + assert rule_id is not None + + result: dict[str, Any] = { + "ruleId": rule_id, + "level": _risk_result_level(finding.bucket, blocking_violation), + "message": { + "text": _risk_result_message(finding, blocking_violation), + }, + "locations": [_file_location(after_path, base_dir)], + "partialFingerprints": { + "ruleId": rule_id, + "componentKey": finding.component_key, + }, + "properties": { + "component_key": finding.component_key, + "component_name": finding.component.name, + "finding_bucket": finding.bucket.value, + "policy_blocking": blocking_violation is not None, + "result_kind": "risk_finding", + }, + } + if blocking_violation is not None: + result["properties"]["blocking_rule_id"] = blocking_violation.rule_id + return result + + +def _policy_violation_to_result( + violation: PolicyViolation, + *, + after_path: Path, + base_dir: Path | None, +) -> dict[str, Any]: + rule_id = sarif_rule_id_for_policy_violation(violation.rule_id) + assert rule_id is not None + + return { + "ruleId": rule_id, + "level": "error", + "message": { + "text": _policy_result_message(violation), + }, + "locations": [_file_location(after_path, base_dir)], + "partialFingerprints": { + "ruleId": rule_id, + "componentKey": violation.component_key or "global-policy-check", + }, + "properties": { + "policy_rule_id": violation.rule_id, + "component_key": violation.component_key, + "component_name": violation.component_name, + "result_kind": "policy_violation", + }, + } + + +def _risk_result_level(bucket: RiskBucket, blocking_violation: PolicyViolation | None) -> str: + if blocking_violation is not None: + return "error" + if bucket is RiskBucket.MAJOR_UPGRADE: + return "note" + return "warning" + + +def _risk_result_message(finding: RiskFinding, blocking_violation: PolicyViolation | None) -> str: + component_label = _component_label(finding.component.name, finding.component.version) + if finding.bucket is RiskBucket.UNKNOWN_LICENSE: + base_message = f"{component_label} has missing or unknown license metadata." + elif finding.bucket is RiskBucket.SUSPICIOUS_SOURCE: + base_message = f"{component_label} has suspicious or incomplete source provenance." + elif finding.bucket is RiskBucket.MAJOR_UPGRADE: + base_message = finding.rationale + else: + base_message = finding.rationale + + if blocking_violation is None: + return base_message + return f"Blocked by policy: {base_message}" + + +def _policy_result_message(violation: PolicyViolation) -> str: + if violation.rule_id == "max_added_packages": + return violation.message + if violation.rule_id == "allow_sources" and violation.component_name: + component_label = _component_label(violation.component_name, None) + return f"{component_label}: {violation.message}" + return violation.message + + +def _component_label(name: str, version: str | None) -> str: + if version: + return f"{name} {version}" + return name + + +def _blocking_violation_map(violations: list[PolicyViolation]) -> dict[tuple[str, str | None], PolicyViolation]: + return { + (violation.rule_id, violation.component_key): violation + for violation in violations + } + + +def _policy_rule_id_for_bucket(bucket: RiskBucket) -> str: + return bucket.value + + +def _file_location(path: Path, base_dir: Path | None) -> dict[str, Any]: + return { + "physicalLocation": { + "artifactLocation": _artifact_location(path, base_dir), + "region": { + "startLine": 1, + }, + } + } + + +def _artifact_location(path: Path, base_dir: Path | None) -> dict[str, str]: + resolved = path.resolve() + if base_dir is not None: + try: + relative = resolved.relative_to(base_dir) + except ValueError: + pass + else: + return { + "uri": relative.as_posix(), + "uriBaseId": "%SRCROOT%", + } + return { + "uri": resolved.as_uri(), + } + + +def _directory_uri(path: Path) -> str: + uri = path.resolve().as_uri() + return uri if uri.endswith("/") else f"{uri}/" + + +def _guardrail_metadata_to_dict(metadata: SarifRenderMetadata) -> dict[str, Any]: + return { + "result_limit": metadata.result_limit, + "total_candidate_results": metadata.total_candidate_results, + "emitted_results": metadata.emitted_results, + "omitted_results": metadata.omitted_results, + "truncated": metadata.truncated, + "prioritization": metadata.prioritization, + "warning": metadata.warning_message, + } + + +def _result_sort_key(result: dict[str, Any]) -> tuple[int, int, int, str, str, str]: + properties = result.get("properties", {}) + if not isinstance(properties, dict): + properties = {} + + result_kind = properties.get("result_kind") + if result_kind == "risk_finding": + kind_rank = 0 + elif properties.get("component_key"): + kind_rank = 1 + else: + kind_rank = 2 + + return ( + _LEVEL_PRIORITY.get(str(result.get("level", "warning")), 99), + kind_rank, + _RULE_PRIORITY.get(str(result.get("ruleId")), 99), + str(properties.get("component_key") or ""), + str(properties.get("component_name") or ""), + str(result.get("ruleId") or ""), + ) + + +def _sarif_rule_metadata(rule_id: str) -> dict[str, Any]: + catalog = rule_catalog_to_dict() + if rule_id.startswith("sdr.policy_violation."): + policy_rule_id = rule_id.removeprefix("sdr.policy_violation.") + description = catalog.get(policy_rule_id, {}).get("description", "Blocking policy violation.") + return { + "id": rule_id, + "name": f"policy_violation.{policy_rule_id}", + "shortDescription": { + "text": f"Blocking policy violation: {policy_rule_id}", + }, + "fullDescription": { + "text": description, + }, + "defaultConfiguration": { + "level": "error", + }, + "properties": { + "tags": ["supply-chain", "policy"], + }, + } + + base_rule_id = rule_id.removeprefix("sdr.") + description = catalog.get(base_rule_id, {}).get("description", base_rule_id) + return { + "id": rule_id, + "name": base_rule_id, + "shortDescription": { + "text": description, + }, + "fullDescription": { + "text": description, + }, + "defaultConfiguration": { + "level": "note" if base_rule_id == "major_upgrade" else "warning", + }, + "properties": { + "tags": ["supply-chain", "sbom"], + }, + } diff --git a/tools/sbom-diff-and-risk/tests/test_cli_exit_codes.py b/tools/sbom-diff-and-risk/tests/test_cli_exit_codes.py index 851f319..12a3424 100644 --- a/tools/sbom-diff-and-risk/tests/test_cli_exit_codes.py +++ b/tools/sbom-diff-and-risk/tests/test_cli_exit_codes.py @@ -1,11 +1,12 @@ from __future__ import annotations +import os import subprocess import sys from pathlib import Path -def test_cli_exit_code_blocking_policy(tmp_path: Path) -> None: +def test_cli_exit_code_blocking_policy_stderr_summary(tmp_path: Path) -> None: project_root = Path(__file__).resolve().parents[1] policy_path = project_root / "examples" / "policy-strict.yml" before = project_root / "examples" / "cdx_before.json" @@ -28,6 +29,9 @@ def test_cli_exit_code_blocking_policy(tmp_path: Path) -> None: ) assert result.returncode == 1 + assert "blocking policy violations detected" in result.stderr + assert "stale_package" in result.stderr + assert "outputs were written" in result.stderr def test_cli_exit_code_warn_only_policy(tmp_path: Path) -> None: @@ -53,6 +57,7 @@ def test_cli_exit_code_warn_only_policy(tmp_path: Path) -> None: ) assert result.returncode == 0 + assert "blocking policy violations detected" not in result.stderr def test_cli_exit_code_invalid_policy_schema(tmp_path: Path) -> None: @@ -77,12 +82,80 @@ def test_cli_exit_code_invalid_policy_schema(tmp_path: Path) -> None: ) assert result.returncode == 2 + assert "Invalid policy schema" in result.stderr + assert "exit code 2" in result.stderr + + +def test_cli_fail_on_flag_blocks(tmp_path: Path) -> None: + project_root = Path(__file__).resolve().parents[1] + before = project_root / "examples" / "cdx_before.json" + after = project_root / "examples" / "cdx_after.json" + + result = _run_compare( + project_root, + [ + "--before", + str(before), + "--after", + str(after), + "--fail-on", + "new_package", + "--warn-on", + "major_upgrade", + "--out-json", + str(tmp_path / "report.json"), + ], + ) + + assert result.returncode == 1 + assert "new_package" in result.stderr + + +def test_cli_compare_help_mentions_policy_flags_and_exit_codes() -> None: + project_root = Path(__file__).resolve().parents[1] + + result = _run_compare(project_root, ["--help"]) + + assert result.returncode == 0 + assert "--out-sarif" in result.stdout + assert "--policy" in result.stdout + assert "--fail-on" in result.stdout + assert "--warn-on" in result.stdout + assert "--strict" in result.stdout + assert "Exit codes: 0 = success/no blocking violations" in result.stdout + + +def test_cli_can_write_sarif_only(tmp_path: Path) -> None: + project_root = Path(__file__).resolve().parents[1] + before = project_root / "examples" / "sarif_before.json" + after = project_root / "examples" / "sarif_after.json" + + result = _run_compare( + project_root, + [ + "--before", + str(before), + "--after", + str(after), + "--policy", + str(project_root / "examples" / "policy-strict.yml"), + "--out-sarif", + str(tmp_path / "report.sarif"), + ], + ) + + assert result.returncode == 1 + assert (tmp_path / "report.sarif").is_file() def _run_compare(project_root: Path, args: list[str]) -> subprocess.CompletedProcess[str]: + env = dict(os.environ) + source_path = str(project_root / "src") + env["PYTHONPATH"] = source_path if not env.get("PYTHONPATH") else f"{source_path}{os.pathsep}{env['PYTHONPATH']}" return subprocess.run( [sys.executable, "-m", "sbom_diff_risk.cli", "compare", *args], cwd=project_root, text=True, capture_output=True, + env=env, ) diff --git a/tools/sbom-diff-and-risk/tests/test_reports.py b/tools/sbom-diff-and-risk/tests/test_reports.py index adce546..d718cf4 100644 --- a/tools/sbom-diff-and-risk/tests/test_reports.py +++ b/tools/sbom-diff-and-risk/tests/test_reports.py @@ -1,16 +1,20 @@ from __future__ import annotations +import json from pathlib import Path from sbom_diff_risk.diffing import diff_components from sbom_diff_risk.models import CompareReport, ReportComponents, ReportMetadata, ReportSummary +from sbom_diff_risk.policy_evaluator import evaluate_policy +from sbom_diff_risk.policy_models import PolicyConfig +from sbom_diff_risk.policy_parser import build_policy from sbom_diff_risk.normalize import normalize_input from sbom_diff_risk.report_json import render_report_json from sbom_diff_risk.report_md import render_report_markdown from sbom_diff_risk.risk import evaluate_risks, summarize_risks -def test_report_json_matches_cyclonedx_golden() -> None: +def test_report_json_matches_cyclonedx_golden_pass() -> None: report = _build_report("cdx_before.json", "cdx_after.json") rendered = render_report_json(report) @@ -19,7 +23,7 @@ def test_report_json_matches_cyclonedx_golden() -> None: assert rendered == expected -def test_report_markdown_matches_cyclonedx_golden() -> None: +def test_report_markdown_matches_cyclonedx_golden_pass() -> None: report = _build_report("cdx_before.json", "cdx_after.json") rendered = render_report_markdown(report) @@ -28,6 +32,42 @@ def test_report_markdown_matches_cyclonedx_golden() -> None: assert rendered == expected +def test_report_json_matches_cyclonedx_policy_warn_golden() -> None: + report = _build_report("cdx_before.json", "cdx_after.json", policy_name="policy-minimal.yml") + + rendered = render_report_json(report) + expected = _read_example("sample-policy-warn-report.json") + + assert rendered == expected + + +def test_report_markdown_matches_cyclonedx_policy_warn_golden() -> None: + report = _build_report("cdx_before.json", "cdx_after.json", policy_name="policy-minimal.yml") + + rendered = render_report_markdown(report) + expected = _read_example("sample-policy-warn-report.md") + + assert rendered == expected + + +def test_report_json_matches_cyclonedx_policy_fail_golden() -> None: + report = _build_report("cdx_before.json", "cdx_after.json", policy_name="policy-strict.yml") + + rendered = render_report_json(report) + expected = _read_example("sample-policy-fail-report.json") + + assert rendered == expected + + +def test_report_markdown_matches_cyclonedx_policy_fail_golden() -> None: + report = _build_report("cdx_before.json", "cdx_after.json", policy_name="policy-strict.yml") + + rendered = render_report_markdown(report) + expected = _read_example("sample-policy-fail-report.md") + + assert rendered == expected + + def test_report_json_matches_requirements_golden() -> None: report = _build_report("requirements_before.txt", "requirements_after.txt") @@ -46,7 +86,51 @@ def test_report_markdown_matches_requirements_golden() -> None: assert rendered == expected -def _build_report(before_name: str, after_name: str) -> CompareReport: +def test_report_json_keeps_legacy_sections() -> None: + report = _build_report("cdx_before.json", "cdx_after.json") + + payload = json.loads(render_report_json(report)) + + assert set(payload) >= { + "summary", + "components", + "risks", + "policy_evaluation", + "blocking_findings", + "warning_findings", + "suppressed_findings", + "rule_catalog", + "metadata", + "notes", + } + assert payload["metadata"]["policy_evaluation"] == payload["policy_evaluation"] + + +def test_reports_render_suppressions_when_policy_ignores_findings() -> None: + policy = PolicyConfig( + version=1, + warn_on=("new_package",), + ignore_rules=("new_package",), + ) + report = _build_report("cdx_before.json", "cdx_after.json", policy=policy) + + payload = json.loads(render_report_json(report)) + markdown = render_report_markdown(report) + + assert payload["suppressed_findings"] + assert payload["suppressed_findings"][0]["suppression_reason"] == "ignored_by_policy" + assert "## Suppressions" in markdown + + +def _build_report( + before_name: str, + after_name: str, + *, + policy_name: str | None = None, + policy: PolicyConfig | None = None, + fail_on: str | None = None, + warn_on: str | None = None, +) -> CompareReport: examples = Path(__file__).resolve().parents[1] / "examples" before_path = examples / before_name after_path = examples / after_name @@ -56,6 +140,25 @@ def _build_report(before_name: str, after_name: str) -> CompareReport: added, removed, changed = diff_components(before_components, after_components) risks = evaluate_risks(added, changed, allowlist=["pypi.org", "files.pythonhosted.org", "github.com"]) + + if policy is None: + built_policy, policy_path = build_policy( + policy_path=(Path("examples") / policy_name) if policy_name else None, + fail_on=fail_on, + warn_on=warn_on, + ) + else: + built_policy = policy + policy_path = None + + policy_evaluation = evaluate_policy( + built_policy, + policy_path=policy_path, + added=added, + changed=changed, + findings=risks, + ) + notes = [ "This tool uses heuristic risk classification.", "No network enrichment was performed.", @@ -82,6 +185,7 @@ def _build_report(before_name: str, after_name: str) -> CompareReport: generated_at=None, strict=False, stub=False, + policy_evaluation=policy_evaluation, ), notes=notes, ) diff --git a/tools/sbom-diff-and-risk/tests/test_sarif.py b/tools/sbom-diff-and-risk/tests/test_sarif.py new file mode 100644 index 0000000..5eb5c7b --- /dev/null +++ b/tools/sbom-diff-and-risk/tests/test_sarif.py @@ -0,0 +1,221 @@ +from __future__ import annotations + +import argparse +import json +import re +from pathlib import Path + +from sbom_diff_risk.cli import run_compare +from sbom_diff_risk.diffing import diff_components +from sbom_diff_risk.models import CompareReport, ReportComponents, ReportMetadata, ReportSummary, RiskBucket +from sbom_diff_risk.normalize import normalize_input +from sbom_diff_risk.policy_evaluator import evaluate_policy +from sbom_diff_risk.policy_parser import build_policy +from sbom_diff_risk.report_sarif import ( + render_report_sarif_output, + render_report_sarif, + sarif_rule_id_for_policy_violation, + sarif_rule_id_for_risk_bucket, +) +from sbom_diff_risk.risk import evaluate_risks, summarize_risks + + +def test_render_report_sarif_matches_golden() -> None: + project_root = Path(__file__).resolve().parents[1] + report, before_path, after_path = _build_report( + "sarif_before.json", + "sarif_after.json", + policy_name="policy-strict.yml", + ) + + rendered = render_report_sarif(report, before_path=before_path, after_path=after_path, base_dir=project_root) + expected = (project_root / "examples" / "sample-sarif.sarif").read_text(encoding="utf-8") + + assert _normalize_sarif_golden(rendered) == _normalize_sarif_golden(expected) + + +def test_sarif_rule_ids_are_stable() -> None: + assert sarif_rule_id_for_risk_bucket(RiskBucket.UNKNOWN_LICENSE) == "sdr.unknown_license" + assert sarif_rule_id_for_risk_bucket(RiskBucket.SUSPICIOUS_SOURCE) == "sdr.suspicious_source" + assert sarif_rule_id_for_risk_bucket(RiskBucket.MAJOR_UPGRADE) == "sdr.major_upgrade" + assert sarif_rule_id_for_policy_violation("max_added_packages") == "sdr.policy_violation.max_added_packages" + assert sarif_rule_id_for_policy_violation("allow_sources") == "sdr.policy_violation.allow_sources" + assert sarif_rule_id_for_policy_violation("stale_package") is None + + +def test_sarif_structure_and_mapping_are_github_compatible() -> None: + project_root = Path(__file__).resolve().parents[1] + report, before_path, after_path = _build_report( + "sarif_before.json", + "sarif_after.json", + policy_name="policy-strict.yml", + ) + + payload = json.loads(render_report_sarif(report, before_path=before_path, after_path=after_path, base_dir=project_root)) + + assert payload["version"] == "2.1.0" + assert payload["$schema"].endswith("sarif-2.1.0.json") + assert len(payload["runs"]) == 1 + + run = payload["runs"][0] + assert run["tool"]["driver"]["name"] == "sbom-diff-risk" + assert run["originalUriBaseIds"]["%SRCROOT%"]["uri"].startswith("file:///") + + rules = {rule["id"] for rule in run["tool"]["driver"]["rules"]} + assert rules == { + "sdr.major_upgrade", + "sdr.policy_violation.allow_sources", + "sdr.policy_violation.max_added_packages", + "sdr.suspicious_source", + "sdr.unknown_license", + } + + results = run["results"] + assert [result["ruleId"] for result in results] == [ + "sdr.suspicious_source", + "sdr.unknown_license", + "sdr.policy_violation.allow_sources", + "sdr.policy_violation.max_added_packages", + "sdr.major_upgrade", + ] + assert any(result["level"] == "error" for result in results) + assert all(result["locations"][0]["physicalLocation"]["artifactLocation"]["uri"] == "examples/sarif_after.json" for result in results) + assert results[0]["message"]["text"].startswith("Blocked by policy:") + assert results[2]["message"]["text"].startswith("mystery-lib") + + +def test_sarif_truncation_is_deterministic_and_recorded_in_metadata() -> None: + project_root = Path(__file__).resolve().parents[1] + report, before_path, after_path = _build_report( + "sarif_before.json", + "sarif_after.json", + policy_name="policy-strict.yml", + ) + + first = render_report_sarif_output( + report, + before_path=before_path, + after_path=after_path, + base_dir=project_root, + result_limit=2, + ) + second = render_report_sarif_output( + report, + before_path=before_path, + after_path=after_path, + base_dir=project_root, + result_limit=2, + ) + + assert first.content == second.content + assert first.metadata.truncated is True + assert first.metadata.total_candidate_results == 5 + assert first.metadata.emitted_results == 2 + assert first.metadata.omitted_results == 3 + assert first.metadata.warning_message is not None + + payload = json.loads(first.content) + run = payload["runs"][0] + assert run["properties"]["sbom_diff_risk"]["truncated"] is True + assert run["properties"]["sbom_diff_risk"]["emitted_results"] == 2 + assert [result["ruleId"] for result in run["results"]] == [ + "sdr.suspicious_source", + "sdr.unknown_license", + ] + + +def test_run_compare_emits_stderr_warning_when_sarif_is_truncated( + tmp_path: Path, + monkeypatch, + capsys, +) -> None: + project_root = Path(__file__).resolve().parents[1] + before = project_root / "examples" / "sarif_before.json" + after = project_root / "examples" / "sarif_after.json" + + monkeypatch.setattr("sbom_diff_risk.report_sarif.DEFAULT_SARIF_RESULT_LIMIT", 2) + + exit_code = run_compare( + argparse.Namespace( + before=before, + after=after, + format="auto", + before_format=None, + after_format=None, + out_json=None, + out_md=None, + out_sarif=tmp_path / "report.sarif", + policy=project_root / "examples" / "policy-strict.yml", + fail_on=None, + warn_on=None, + strict=False, + enrich_pypi=False, + source_allowlist="pypi.org,files.pythonhosted.org,github.com", + ) + ) + + stderr = capsys.readouterr().err + assert exit_code == 1 + assert "SARIF results were truncated deterministically" in stderr + assert "limit 2" in stderr + assert (tmp_path / "report.sarif").is_file() + + +def _build_report(before_name: str, after_name: str, *, policy_name: str | None = None) -> tuple[CompareReport, Path, Path]: + project_root = Path(__file__).resolve().parents[1] + examples = project_root / "examples" + before_path = examples / before_name + after_path = examples / after_name + + before_format, before_components, before_notes = normalize_input(before_path) + after_format, after_components, after_notes = normalize_input(after_path) + + added, removed, changed = diff_components(before_components, after_components) + risks = evaluate_risks(added, changed, allowlist=["pypi.org", "files.pythonhosted.org", "github.com"]) + policy, policy_path = build_policy(policy_path=(Path("examples") / policy_name) if policy_name else None) + policy_evaluation = evaluate_policy( + policy, + policy_path=policy_path, + added=added, + changed=changed, + findings=risks, + ) + notes = [ + "This tool uses heuristic risk classification.", + "No network enrichment was performed.", + *before_notes, + *after_notes, + ] + + report = CompareReport( + summary=ReportSummary( + added=len(added), + removed=len(removed), + changed=len(changed), + risk_counts=summarize_risks(risks), + ), + components=ReportComponents( + added=added, + removed=removed, + changed=changed, + ), + risks=risks, + metadata=ReportMetadata( + before_format=before_format, + after_format=after_format, + generated_at=None, + strict=False, + stub=False, + policy_evaluation=policy_evaluation, + ), + notes=notes, + ) + return report, before_path, after_path + + +def _normalize_sarif_golden(value: str) -> str: + return re.sub( + r"file:///[^\"\r\n]+/tools/sbom-diff-and-risk(?:-real)?/", + "file:///__PROJECT_ROOT__/", + value, + ) From 2deaa588b317707d83b1960d4a3228b39b7986f8 Mon Sep 17 00:00:00 2001 From: stacknil Date: Wed, 15 Apr 2026 00:11:24 +0800 Subject: [PATCH 3/6] Add GitHub code scanning workflow example --- .../sbom-diff-and-risk-code-scanning.yml | 43 +++++++++++++ tools/sbom-diff-and-risk/README.md | 2 + .../docs/github-code-scanning.md | 60 +++++++++++++++++++ 3 files changed, 105 insertions(+) create mode 100644 .github/workflows/sbom-diff-and-risk-code-scanning.yml create mode 100644 tools/sbom-diff-and-risk/docs/github-code-scanning.md diff --git a/.github/workflows/sbom-diff-and-risk-code-scanning.yml b/.github/workflows/sbom-diff-and-risk-code-scanning.yml new file mode 100644 index 0000000..a843d83 --- /dev/null +++ b/.github/workflows/sbom-diff-and-risk-code-scanning.yml @@ -0,0 +1,43 @@ +name: sbom-diff-and-risk-code-scanning + +on: + workflow_dispatch: + pull_request: + paths: + - ".github/workflows/sbom-diff-and-risk-code-scanning.yml" + - "tools/sbom-diff-and-risk/**" + +jobs: + upload-sarif: + runs-on: ubuntu-latest + permissions: + security-events: write + contents: read + defaults: + run: + working-directory: tools/sbom-diff-and-risk + steps: + - name: Check out repository + uses: actions/checkout@v5 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: "3.11" + + - name: Install sbom-diff-and-risk + run: python -m pip install -e .[dev] + + - name: Generate SARIF report + run: | + mkdir -p outputs + python -m sbom_diff_risk.cli compare \ + --before examples/sarif_before.json \ + --after examples/sarif_after.json \ + --out-sarif outputs/report.sarif + + - name: Upload SARIF to code scanning + uses: github/codeql-action/upload-sarif@v4 + with: + sarif_file: tools/sbom-diff-and-risk/outputs/report.sarif + category: sbom-diff-risk/example diff --git a/tools/sbom-diff-and-risk/README.md b/tools/sbom-diff-and-risk/README.md index bb8193c..5cdf154 100644 --- a/tools/sbom-diff-and-risk/README.md +++ b/tools/sbom-diff-and-risk/README.md @@ -207,6 +207,8 @@ sbom-diff-risk compare \ --out-sarif outputs/report.sarif ``` +For GitHub code scanning integration guidance and a minimal upload workflow, see [docs/github-code-scanning.md](D:/OneDrive/Code/scientific-computing-toolkit-real/tools/sbom-diff-and-risk/docs/github-code-scanning.md). + ## Limitations - v0.1 is local-file based only. diff --git a/tools/sbom-diff-and-risk/docs/github-code-scanning.md b/tools/sbom-diff-and-risk/docs/github-code-scanning.md new file mode 100644 index 0000000..2f01476 --- /dev/null +++ b/tools/sbom-diff-and-risk/docs/github-code-scanning.md @@ -0,0 +1,60 @@ +# GitHub Code Scanning Integration + +`sbom-diff-and-risk` can export a GitHub-compatible SARIF 2.1.0 subset and upload it with `github/codeql-action/upload-sarif`. + +This project remains local, deterministic, and conservative by default. The GitHub integration is only a transport path for selected high-signal findings. + +## What the example workflow does + +The example workflow in `.github/workflows/sbom-diff-and-risk-code-scanning.yml`: + +- checks out the repository +- installs Python and the local tool +- runs `sbom-diff-risk compare ... --out-sarif` +- uploads the generated SARIF file with `github/codeql-action/upload-sarif` + +The example intentionally uses local example inputs and does not depend on secrets or network enrichment. +It also keeps the compare step at exit code `0` for readability. If you intentionally enforce blocking policy rules during CI and still want SARIF uploaded, add `continue-on-error: true` to the compare step and gate the upload step with `if: always()`. + +## Required permissions + +At minimum, the upload job needs: + +- `security-events: write` + +For private repositories, GitHub also documents `contents: read`. If your workflow needs to inspect other workflow artifacts, `actions: read` may also be required. + +## SARIF guardrails + +GitHub documents both SARIF file-size limits and object-count limits for code scanning uploads. In particular: + +- gzip-compressed SARIF uploads over 10 MB are rejected +- a run may contain up to 25,000 results, but GitHub only includes the top 5,000 results for display, prioritized by severity + +To keep uploads reviewable and GitHub-oriented, `sbom-diff-risk` applies a deterministic SARIF result cap of 5,000 results. When truncation happens: + +- results are prioritized as `error`, then `warning`, then `note` +- direct mapped findings are kept ahead of policy-only checks +- stable tie-breakers are applied by rule ID and component identity +- truncation is recorded in SARIF run metadata +- the CLI emits a warning to stderr + +This does not guarantee every huge SARIF file will fit under GitHub's documented upload-size limits, but it prevents silent overproduction of low-priority results. + +## When to use a SARIF category + +Set a SARIF category when you upload more than one analysis for the same commit and tool. Common cases include: + +- one upload per manifest type +- one upload per monorepo slice +- separate policy modes or rule packs + +If you upload multiple SARIF files for the same tool and commit without distinct categories, later uploads replace earlier ones. In GitHub Actions, set the `category:` input on `github/codeql-action/upload-sarif`. Outside Actions, use `runAutomationDetails.id` in the SARIF file. + +## What this integration does not cover + +- It does not add CVE lookup or advisory enrichment. +- It does not make exact line mappings for manifests that do not expose stable locations. +- It does not automatically handle every possible multi-workflow or monorepo routing strategy. +- It does not package `sbom-diff-risk` as a GitHub Marketplace Action. +- It does not bypass GitHub's documented SARIF ingestion limits. From 438e97076914c46af54394c1c9f7cc8d9c943b3f Mon Sep 17 00:00:00 2001 From: stacknil Date: Wed, 15 Apr 2026 00:12:57 +0800 Subject: [PATCH 4/6] Tighten parser boundaries for deterministic inputs --- tools/sbom-diff-and-risk/README.md | 77 ++++++++--- .../docs/parser-boundaries.md | 49 +++++++ tools/sbom-diff-and-risk/docs/sbom-basics.md | 10 +- .../examples/pyproject_groups_after.toml | 25 ++++ .../examples/pyproject_groups_before.toml | 24 ++++ .../src/sbom_diff_risk/cli.py | 23 +++- .../src/sbom_diff_risk/errors.py | 14 +- .../src/sbom_diff_risk/normalize.py | 12 ++ .../src/sbom_diff_risk/parsers/common.py | 20 +-- .../parsers/pyproject_groups.py | 121 ++++++++++++++++++ .../sbom_diff_risk/parsers/pyproject_toml.py | 59 +++++++-- .../parsers/requirements_rules.py | 62 +++++++++ .../parsers/requirements_txt.py | 19 ++- .../fixtures/pyproject_groups_after.toml | 25 ++++ .../fixtures/pyproject_groups_before.toml | 24 ++++ .../tests/fixtures/pyproject_parser.toml | 2 +- .../tests/fixtures/requirements_parser.txt | 8 +- .../tests/test_cli_exit_codes.py | 75 +++++++++++ .../tests/test_normalize.py | 28 +++- .../sbom-diff-and-risk/tests/test_parsers.py | 113 +++++++++++++--- 20 files changed, 713 insertions(+), 77 deletions(-) create mode 100644 tools/sbom-diff-and-risk/docs/parser-boundaries.md create mode 100644 tools/sbom-diff-and-risk/examples/pyproject_groups_after.toml create mode 100644 tools/sbom-diff-and-risk/examples/pyproject_groups_before.toml create mode 100644 tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/pyproject_groups.py create mode 100644 tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/requirements_rules.py create mode 100644 tools/sbom-diff-and-risk/tests/fixtures/pyproject_groups_after.toml create mode 100644 tools/sbom-diff-and-risk/tests/fixtures/pyproject_groups_before.toml diff --git a/tools/sbom-diff-and-risk/README.md b/tools/sbom-diff-and-risk/README.md index 5cdf154..2abf653 100644 --- a/tools/sbom-diff-and-risk/README.md +++ b/tools/sbom-diff-and-risk/README.md @@ -9,6 +9,7 @@ It uses conservative heuristics for change intelligence. By default it does not - Normalize two local inputs into a shared component schema. - Diff components as `added`, `removed`, and `changed`. - Apply conservative, heuristic risk buckets to newly added and changed components. +- Apply optional local policy enforcement over those findings. - Produce machine-friendly JSON and reviewer-friendly Markdown reports. - Stay fully local-file based by default. @@ -42,13 +43,15 @@ When a `purl` includes a version, the tool keeps the full value in `Component.pu - No reputation scoring or malware verdicts. - No hidden enrichment or implicit network access. - No web UI. +- No packaged GitHub Marketplace Action. ## Supported Formats - CycloneDX JSON - SPDX JSON - `requirements.txt` -- `pyproject.toml` +- `pyproject.toml` via PEP 621 `[project]` metadata +- `pyproject.toml` dependency groups via PEP 735 `[dependency-groups]` with explicit selection ## Risk Bucket Semantics @@ -123,6 +126,18 @@ sbom-diff-risk compare \ --out-md outputs/pyproject-report.md ``` +Generate reports for a specific PEP 735 dependency group: + +```bash +sbom-diff-risk compare \ + --before examples/pyproject_groups_before.toml \ + --after examples/pyproject_groups_after.toml \ + --format pyproject-toml \ + --pyproject-group dev \ + --out-json outputs/pyproject-groups-report.json \ + --out-md outputs/pyproject-groups-report.md +``` + ## CLI Flags - `--before path` @@ -130,6 +145,7 @@ sbom-diff-risk compare \ - `--format auto|cyclonedx-json|spdx-json|requirements-txt|pyproject-toml` - `--before-format cyclonedx-json|spdx-json|requirements-txt|pyproject-toml` - `--after-format cyclonedx-json|spdx-json|requirements-txt|pyproject-toml` +- `--pyproject-group name` - `--out-json path` - `--out-md path` - `--out-sarif path` @@ -144,18 +160,19 @@ sbom-diff-risk compare \ ## Examples -The [`examples/`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples) directory includes: +The [examples/](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples) directory includes: - before/after inputs for CycloneDX JSON, SPDX JSON, `requirements.txt`, and `pyproject.toml` +- dependency-group examples at `examples/pyproject_groups_before.toml` and `examples/pyproject_groups_after.toml` - example policies at `examples/policy-minimal.yml` and `examples/policy-strict.yml` -- a sample CycloneDX-based JSON report at [`sample-report.json`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-report.json) -- a sample CycloneDX-based Markdown report at [`sample-report.md`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-report.md) -- sample policy-warn reports at [`sample-policy-warn-report.json`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.json) and [`sample-policy-warn-report.md`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.md) -- sample policy-fail reports at [`sample-policy-fail-report.json`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.json) and [`sample-policy-fail-report.md`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.md) -- a sample SARIF export at [`sample-sarif.sarif`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-sarif.sarif) -- requirements-based sample reports at [`sample-requirements-report.json`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-requirements-report.json) and [`sample-requirements-report.md`](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-requirements-report.md) +- a sample pass JSON report at [sample-report.json](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-report.json) +- a sample pass Markdown report at [sample-report.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-report.md) +- sample policy-warn reports at [sample-policy-warn-report.json](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.json) and [sample-policy-warn-report.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.md) +- sample policy-fail reports at [sample-policy-fail-report.json](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.json) and [sample-policy-fail-report.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.md) +- a sample SARIF export at [sample-sarif.sarif](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-sarif.sarif) +- requirements-based sample reports at [sample-requirements-report.json](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-requirements-report.json) and [sample-requirements-report.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/examples/sample-requirements-report.md) -## Enforcement +## Enforcement Mode Policy enforcement is optional and deterministic. Exit codes are stable: @@ -207,23 +224,47 @@ sbom-diff-risk compare \ --out-sarif outputs/report.sarif ``` -For GitHub code scanning integration guidance and a minimal upload workflow, see [docs/github-code-scanning.md](D:/OneDrive/Code/scientific-computing-toolkit-real/tools/sbom-diff-and-risk/docs/github-code-scanning.md). +For GitHub code scanning integration guidance and a minimal upload workflow, see [docs/github-code-scanning.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/docs/github-code-scanning.md). + +## Parser Boundaries + +Deterministic local mode intentionally supports a conservative subset of packaging syntax. The detailed matrix lives in [docs/parser-boundaries.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/docs/parser-boundaries.md). + +### requirements.txt subset + +| Syntax | Status | Notes | +| --- | --- | --- | +| Plain PEP 508 requirement entries | Supported | Names, specifiers, extras, and markers | +| Comments, blank lines, line continuations | Supported | Normalized locally without installer behavior | +| `-r`, `--requirement` | Unsupported | Include chains fail closed | +| `-c`, `--constraint` | Unsupported | Constraint files fail closed | +| Editable installs | Unsupported | `-e` and `--editable` are rejected | +| Direct URL, VCS, and local path refs | Unsupported | Includes `pkg @ https://...`, `git+...`, wheels, archives, and local paths | +| Index and source options | Unsupported | Includes `--index-url`, `--extra-index-url`, `--find-links`, and related flags | + +### pyproject.toml subset + +- default parsing supports PEP 621 `[project.dependencies]` and `[project.optional-dependencies]` +- dependency groups are supported through PEP 735 `[dependency-groups]` +- dependency groups must be selected explicitly with `--pyproject-group ` +- dependency groups are not treated as aliases for `[project.optional-dependencies]` +- tool-specific layouts such as Poetry, Hatch, and PDM remain out of scope in v0.2 ## Limitations -- v0.1 is local-file based only. +- default mode is local-file based only. - `generated_at` remains `null` to preserve deterministic report output. - `stale_package` is not resolved offline. The report emits `not_evaluated` instead. -- SARIF export intentionally covers only a conservative subset of findings in v0.1. +- SARIF export intentionally covers only a conservative subset of findings in v0.2. - No vulnerability database integration, CVE matching, or advisory enrichment. -- `requirements.txt` support intentionally covers a conservative subset: plain PEP 508 requirement entries, comments, direct URL requirements, and line continuations. -- `requirements.txt` intentionally does not support pip include/constraint directives such as `-r`, `-c`, or arbitrary install flags in v0.1. -- `pyproject.toml` support intentionally covers a conservative subset: PEP 621 `[project.dependencies]` and `[project.optional-dependencies]`. -- `pyproject.toml` intentionally does not support tool-specific layouts such as Poetry, Hatch, or PDM sections in v0.1. +- `requirements.txt` support intentionally covers a conservative subset: plain PEP 508 requirement entries, comments, extras, markers, and line continuations. +- `requirements.txt` intentionally rejects include/constraint directives, editable installs, direct URL/path refs, index/source options, and other pip-only install flags in deterministic mode. +- `pyproject.toml` support intentionally covers a conservative subset: PEP 621 `[project.dependencies]`, `[project.optional-dependencies]`, and explicit PEP 735 `[dependency-groups]` selection. +- `pyproject.toml` intentionally does not support tool-specific layouts such as Poetry, Hatch, or PDM sections in v0.2. - Risk buckets are heuristics, not security verdicts. - Runtime-generated `outputs/` artifacts are ignored; tracked examples live in `examples/`. -- Policy files are YAML-only in v0.1 and unknown rule ids fail closed. +- Policy files are YAML-only in v0.2 and unknown rule ids fail closed. ## Current Status -The project now normalizes local CycloneDX JSON, SPDX JSON, `requirements.txt`, and PEP 621 `pyproject.toml` inputs into the shared component model, diffs them deterministically, and generates stable JSON/Markdown/SARIF reports with golden tests and optional policy enforcement. +The project now normalizes local CycloneDX JSON, SPDX JSON, `requirements.txt`, and conservative `pyproject.toml` inputs, including explicit PEP 735 dependency-group selection, into the shared component model, diffs them deterministically, and generates stable JSON/Markdown/SARIF reports with tests and optional policy enforcement. diff --git a/tools/sbom-diff-and-risk/docs/parser-boundaries.md b/tools/sbom-diff-and-risk/docs/parser-boundaries.md new file mode 100644 index 0000000..2d289c0 --- /dev/null +++ b/tools/sbom-diff-and-risk/docs/parser-boundaries.md @@ -0,0 +1,49 @@ +# Parser boundaries + +`sbom-diff-and-risk` intentionally supports a conservative parser subset so local runs remain deterministic, auditable, and CI-friendly. + +The project does not try to emulate a package installer. When syntax would require resolver behavior, implicit includes, index lookups, or environment-specific side effects, the parser fails closed with an explicit error. + +## requirements.txt + +`requirements.txt` is treated as a narrow manifest format, not as "everything pip can do in a file". + +| Syntax | Status | Notes | +| --- | --- | --- | +| Plain PEP 508 names and version specifiers | Supported | Example: `requests==2.31.0` | +| Extras and markers | Supported | Example: `pytest[testing]>=8.0 ; python_version >= "3.11"` | +| Comments and blank lines | Supported | Stripped before parsing | +| Line continuations | Supported | Continued lines are joined deterministically | +| `-r`, `--requirement` | Unsupported | Include chains are rejected | +| `-c`, `--constraint` | Unsupported | Constraint files are rejected | +| `-e`, `--editable` | Unsupported | Editable installs are rejected | +| Direct URL, VCS, or local path references | Unsupported | Includes `pkg @ https://...`, `git+...`, `file://...`, wheels, and local archives | +| Index and source options | Unsupported | Includes `--index-url`, `--extra-index-url`, `--find-links`, `--trusted-host`, `--no-index` | +| Other pip-only install flags | Unsupported | Includes hash flags, binary toggles, prerelease flags, and related installer controls | + +When unsupported syntax appears, the parser raises `UnsupportedInputError` and the CLI returns exit code `2`. + +## pyproject.toml + +`pyproject.toml` support is also intentionally narrow: + +| Section | Status | Notes | +| --- | --- | --- | +| `[project.dependencies]` | Supported | Parsed by default | +| `[project.optional-dependencies]` | Supported | Parsed by default and kept distinct from dependency groups | +| `[dependency-groups]` | Supported | Requires explicit `--pyproject-group ` selection | +| `{ include-group = "name" }` inside dependency groups | Supported | Includes are resolved locally and deterministically | +| Missing requested dependency group | Explicit error | Reported as `InputSelectionError` | +| Poetry, Hatch, PDM, or other tool-specific dependency sections | Unsupported | Not parsed in v0.2 | + +Dependency groups are not merged automatically with `[project.optional-dependencies]`. They solve different problems and are kept separate on purpose. + +## Error taxonomy + +The parser uses explicit error classes so CI logs are understandable: + +- `MalformedInputError`: the file is syntactically malformed. +- `UnsupportedInputError`: the file is valid enough to read, but deterministic mode intentionally does not support the construct. +- `InputSelectionError`: the user asked for a parser selection the input cannot satisfy, such as a missing dependency group. + +The CLI maps these parser failures to exit code `2`. diff --git a/tools/sbom-diff-and-risk/docs/sbom-basics.md b/tools/sbom-diff-and-risk/docs/sbom-basics.md index 0f174b2..04f64d9 100644 --- a/tools/sbom-diff-and-risk/docs/sbom-basics.md +++ b/tools/sbom-diff-and-risk/docs/sbom-basics.md @@ -2,7 +2,7 @@ This project treats SBOMs as one possible source of dependency inventory data. -For v0.1, the tool is intentionally limited to local-file parsing, normalization, diffing, and heuristic reporting. +For v0.2, the tool is intentionally limited to local-file parsing, normalization, diffing, and conservative heuristic reporting. ## Supported local inputs @@ -17,18 +17,20 @@ For v0.1, the tool is intentionally limited to local-file parsing, normalization - supported: plain PEP 508 requirement entries - supported: comments and blank lines -- supported: direct URL requirements - supported: line continuations -- not supported: `-r`, `--requirement`, `-c`, `--constraint`, or arbitrary pip install flags +- not supported: `-r`, `--requirement`, `-c`, `--constraint`, editable installs, direct URL/path refs, or pip index/options -`pyproject.toml` support is intentionally conservative in v0.1: +`pyproject.toml` support is intentionally conservative in v0.2: - supported: PEP 621 `[project.dependencies]` - supported: PEP 621 `[project.optional-dependencies]` +- supported: PEP 735 `[dependency-groups]` with explicit `--pyproject-group` selection - not supported: Poetry, Hatch, PDM, or other tool-specific dependency sections These boundaries are deliberate so the tool can stay deterministic and explicit about what it does and does not parse. +For the detailed supported/unsupported matrix, see [parser-boundaries.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/docs/parser-boundaries.md). + ## Normalization goals - keep one internal `Component` model diff --git a/tools/sbom-diff-and-risk/examples/pyproject_groups_after.toml b/tools/sbom-diff-and-risk/examples/pyproject_groups_after.toml new file mode 100644 index 0000000..e91aea6 --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/pyproject_groups_after.toml @@ -0,0 +1,25 @@ +[project] +name = "grouped-project" +version = "0.2.0" +dependencies = [ + "requests==2.32.0", +] + +[project.optional-dependencies] +docs = [ + "mkdocs>=1.6", +] + +[dependency-groups] +lint = [ + "ruff==0.6.3", +] +dev = [ + "pytest==8.3.1", + "mypy==1.11.2", + { include-group = "lint" }, +] +test = [ + "pytest-cov==5.0.0", + { include-group = "dev" }, +] diff --git a/tools/sbom-diff-and-risk/examples/pyproject_groups_before.toml b/tools/sbom-diff-and-risk/examples/pyproject_groups_before.toml new file mode 100644 index 0000000..67d38bd --- /dev/null +++ b/tools/sbom-diff-and-risk/examples/pyproject_groups_before.toml @@ -0,0 +1,24 @@ +[project] +name = "grouped-project" +version = "0.1.0" +dependencies = [ + "requests==2.31.0", +] + +[project.optional-dependencies] +docs = [ + "mkdocs>=1.6", +] + +[dependency-groups] +lint = [ + "ruff==0.5.0", +] +dev = [ + "pytest==8.2.0", + { include-group = "lint" }, +] +test = [ + "pytest-cov==5.0.0", + { include-group = "dev" }, +] diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/cli.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/cli.py index 41aa8ec..8faf99f 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/cli.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/cli.py @@ -8,7 +8,7 @@ from .diffing import diff_components from .errors import ParseError, PolicyError from .models import CompareReport, ReportComponents, ReportMetadata, ReportSummary -from .normalize import SUPPORTED_FORMATS, normalize_input +from .normalize import SUPPORTED_FORMATS, normalize_input_with_options from .policy_evaluator import evaluate_policy from .policy_parser import build_policy from .presentation import effective_policy_evaluation, summarize_violations_by_rule @@ -51,6 +51,11 @@ def build_parser() -> argparse.ArgumentParser: default=None, help="Explicit format for the after input.", ) + compare.add_argument( + "--pyproject-group", + default=None, + help="Select a PEP 735 [dependency-groups] group when a compared input is pyproject.toml.", + ) compare.add_argument("--out-json", type=Path, default=None, help="Write a JSON report to this path.") compare.add_argument("--out-md", type=Path, default=None, help="Write a Markdown report to this path.") compare.add_argument( @@ -107,6 +112,8 @@ def main(argv: Sequence[str] | None = None) -> int: def run_compare(args: argparse.Namespace) -> int: + pyproject_group = getattr(args, "pyproject_group", None) + if args.enrich_pypi: raise NotImplementedError("--enrich-pypi is reserved for a later network-enabled release.") @@ -122,8 +129,18 @@ def run_compare(args: argparse.Namespace) -> int: before_declared = _resolve_declared_format(args.format, args.before_format) after_declared = _resolve_declared_format(args.format, args.after_format) - before_format, before_components, before_notes = normalize_input(before_path, before_declared) - after_format, after_components, after_notes = normalize_input(after_path, after_declared) + before_format, before_components, before_notes = normalize_input_with_options( + before_path, + declared_format=before_declared, + pyproject_group=pyproject_group, + ) + after_format, after_components, after_notes = normalize_input_with_options( + after_path, + declared_format=after_declared, + pyproject_group=pyproject_group, + ) + if pyproject_group and before_format != "pyproject-toml" and after_format != "pyproject-toml": + raise ValueError("--pyproject-group requires at least one pyproject.toml input.") policy, policy_path = build_policy(policy_path=args.policy, fail_on=args.fail_on, warn_on=args.warn_on) added, removed, changed = diff_components(before_components, after_components) diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/errors.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/errors.py index 6f8f189..01b9230 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/errors.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/errors.py @@ -5,5 +5,17 @@ class ParseError(ValueError): """Raised when an input file cannot be parsed into normalized components.""" +class MalformedInputError(ParseError): + """Raised when an input is syntactically malformed.""" + + +class UnsupportedInputError(ParseError): + """Raised when deterministic mode rejects otherwise valid input syntax.""" + + +class InputSelectionError(ParseError): + """Raised when an explicit parser selection cannot be satisfied.""" + + class PolicyError(ValueError): - """Raised when a policy file or policy override is invalid.""" + """Raised when policy parsing or evaluation inputs are invalid.""" diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/normalize.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/normalize.py index 759f95a..13d4c35 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/normalize.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/normalize.py @@ -47,11 +47,23 @@ def detect_format(path: Path) -> str: def normalize_input(path: Path, declared_format: str | None = None) -> tuple[str, list[Component], list[str]]: + return normalize_input_with_options(path, declared_format=declared_format, pyproject_group=None) + + +def normalize_input_with_options( + path: Path, + *, + declared_format: str | None = None, + pyproject_group: str | None = None, +) -> tuple[str, list[Component], list[str]]: selected_format = declared_format or detect_format(path) if selected_format not in SUPPORTED_FORMATS: raise ValueError( f"Unsupported input format {selected_format!r}. Supported formats: {', '.join(SUPPORTED_FORMATS)}." ) + if selected_format == "pyproject-toml": + return selected_format, pyproject_toml.parse(path, dependency_group=pyproject_group), [] + parser = _FORMAT_PARSERS[selected_format] return selected_format, parser(path), [] diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/common.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/common.py index 0ed8051..b8a9e1e 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/common.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/common.py @@ -9,19 +9,19 @@ from packaging.requirements import InvalidRequirement, Requirement from packaging.utils import canonicalize_name -from ..errors import ParseError +from ..errors import MalformedInputError def load_json_object(path: Path, format_name: str) -> dict[str, Any]: try: payload = json.loads(path.read_text(encoding="utf-8")) except json.JSONDecodeError as exc: - raise ParseError( + raise MalformedInputError( f"Malformed {format_name} JSON in {path} at line {exc.lineno}, column {exc.colno}: {exc.msg}." ) from exc if not isinstance(payload, dict): - raise ParseError(f"Malformed {format_name} input in {path}: top-level JSON value must be an object.") + raise MalformedInputError(f"Malformed {format_name} input in {path}: top-level JSON value must be an object.") return payload @@ -29,22 +29,22 @@ def load_toml_object(path: Path, format_name: str) -> dict[str, Any]: try: payload = tomllib.loads(path.read_text(encoding="utf-8")) except tomllib.TOMLDecodeError as exc: - raise ParseError(f"Malformed {format_name} TOML in {path}: {exc}.") from exc + raise MalformedInputError(f"Malformed {format_name} TOML in {path}: {exc}.") from exc if not isinstance(payload, dict): - raise ParseError(f"Malformed {format_name} input in {path}: top-level TOML value must be a table.") + raise MalformedInputError(f"Malformed {format_name} input in {path}: top-level TOML value must be a table.") return payload def require_mapping(value: Any, context: str) -> dict[str, Any]: if not isinstance(value, dict): - raise ParseError(f"Malformed input: expected object for {context}.") + raise MalformedInputError(f"Malformed input: expected object for {context}.") return value def require_list(value: Any, context: str) -> list[Any]: if not isinstance(value, list): - raise ParseError(f"Malformed input: expected array for {context}.") + raise MalformedInputError(f"Malformed input: expected array for {context}.") return value @@ -52,7 +52,7 @@ def optional_str(value: Any, context: str) -> str | None: if value is None: return None if not isinstance(value, str): - raise ParseError(f"Malformed input: expected string for {context}.") + raise MalformedInputError(f"Malformed input: expected string for {context}.") stripped = value.strip() return stripped or None @@ -60,7 +60,7 @@ def optional_str(value: Any, context: str) -> str | None: def required_str(value: Any, context: str) -> str: parsed = optional_str(value, context) if parsed is None: - raise ParseError(f"Malformed input: missing required string for {context}.") + raise MalformedInputError(f"Malformed input: missing required string for {context}.") return parsed @@ -95,7 +95,7 @@ def parse_requirement_text(raw_requirement: str, source_description: str) -> Req try: return Requirement(raw_requirement) except InvalidRequirement as exc: - raise ParseError(f"Malformed requirement in {source_description}: {raw_requirement!r}. {exc}") from exc + raise MalformedInputError(f"Malformed requirement in {source_description}: {raw_requirement!r}. {exc}") from exc def extract_requirement_version(requirement: Requirement) -> tuple[str | None, str | None]: diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/pyproject_groups.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/pyproject_groups.py new file mode 100644 index 0000000..01bc3d8 --- /dev/null +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/pyproject_groups.py @@ -0,0 +1,121 @@ +from __future__ import annotations + +import re +from collections import defaultdict +from typing import Any + +from ..errors import InputSelectionError, MalformedInputError, UnsupportedInputError +from .common import require_list, require_mapping + +_GROUP_NAME_NORMALIZE_RE = re.compile(r"[-_.]+") + + +def normalize_group_name(name: str) -> str: + return _GROUP_NAME_NORMALIZE_RE.sub("-", name).lower() + + +def normalize_dependency_groups(raw_groups: object, context: str) -> tuple[dict[str, list[Any]], dict[str, str]]: + if raw_groups is None: + return {}, {} + + groups = require_mapping(raw_groups, context) + normalized_groups: dict[str, list[Any]] = {} + original_names: dict[str, str] = {} + collisions: defaultdict[str, list[str]] = defaultdict(list) + + for group_name, raw_value in groups.items(): + if not isinstance(group_name, str): + raise MalformedInputError(f"Malformed pyproject.toml in {context}: dependency group names must be strings.") + normalized_name = normalize_group_name(group_name) + collisions[normalized_name].append(group_name) + normalized_groups[normalized_name] = require_list(raw_value, f"{context}.{group_name}") + original_names[normalized_name] = group_name + + duplicates = [f"{normalized} ({', '.join(names)})" for normalized, names in collisions.items() if len(names) > 1] + if duplicates: + raise InputSelectionError( + "Duplicate dependency group names after normalization: " + ", ".join(sorted(duplicates)) + "." + ) + + return normalized_groups, original_names + + +def resolve_dependency_group( + dependency_groups: dict[str, list[Any]], + original_names: dict[str, str], + *, + requested_group: str, + context: str, +) -> tuple[str, list[str]]: + normalized_requested_group = normalize_group_name(requested_group) + if normalized_requested_group not in dependency_groups: + raise InputSelectionError( + f"Requested dependency group {requested_group!r} was not found in [dependency-groups] of {context}. " + "Dependency groups are distinct from [project.optional-dependencies]." + ) + + resolved = _resolve_group( + dependency_groups, + original_names, + group=normalized_requested_group, + context=context, + past_groups=(), + ) + return original_names[normalized_requested_group], resolved + + +def _resolve_group( + dependency_groups: dict[str, list[Any]], + original_names: dict[str, str], + *, + group: str, + context: str, + past_groups: tuple[str, ...], +) -> list[str]: + if group in past_groups: + cycle = " -> ".join([*(original_names[item] for item in past_groups), original_names[group]]) + raise UnsupportedInputError(f"Cyclic dependency group include in {context}: {cycle}.") + + raw_group = dependency_groups[group] + realized_group: list[str] = [] + for index, item in enumerate(raw_group, start=1): + if isinstance(item, str): + realized_group.append(item) + continue + + if isinstance(item, dict): + if tuple(item.keys()) != ("include-group",): + raise UnsupportedInputError( + f"Unsupported dependency group item in {context}:{original_names[group]}[{index}]: {item!r}. " + "Only strings and {include-group = \"name\"} objects are supported." + ) + + include_name = item["include-group"] + if not isinstance(include_name, str) or not include_name.strip(): + raise MalformedInputError( + f"Malformed dependency group include in {context}:{original_names[group]}[{index}]: " + "include-group must be a non-empty string." + ) + normalized_include = normalize_group_name(include_name) + if normalized_include not in dependency_groups: + raise InputSelectionError( + f"Dependency group include {include_name!r} referenced by {original_names[group]!r} " + f"was not found in [dependency-groups] of {context}." + ) + realized_group.extend( + _resolve_group( + dependency_groups, + original_names, + group=normalized_include, + context=context, + past_groups=(*past_groups, group), + ) + ) + continue + + raise MalformedInputError( + f"Malformed dependency group item in {context}:{original_names[group]}[{index}]: " + "items must be strings or include-group objects." + ) + + return realized_group diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/pyproject_toml.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/pyproject_toml.py index df6d078..7984925 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/pyproject_toml.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/pyproject_toml.py @@ -2,22 +2,38 @@ from pathlib import Path -from ..errors import ParseError +from ..errors import InputSelectionError, MalformedInputError, UnsupportedInputError from ..models import Component from .common import build_pypi_purl, extract_requirement_version, load_toml_object, parse_requirement_text, require_mapping +from .pyproject_groups import normalize_dependency_groups, resolve_dependency_group -def parse(path: Path) -> list[Component]: +def parse(path: Path, *, dependency_group: str | None = None) -> list[Component]: payload = load_toml_object(path, "pyproject") + if dependency_group is not None: + return _parse_dependency_group(path, payload, dependency_group) + raw_project = payload.get("project") if raw_project is None: - raise ParseError( + if payload.get("dependency-groups") is not None: + raise InputSelectionError( + f"pyproject.toml in {path} defines [dependency-groups]; select one explicitly with --pyproject-group." + ) + raise UnsupportedInputError( f"Unsupported pyproject.toml layout in {path}: only PEP 621 [project] dependencies are supported." ) project = require_mapping(raw_project, f"{path}: project") normalized: list[Component] = [] - normalized.extend(_parse_requirement_group(path, project.get("dependencies"), "dependencies", "project-dependency")) + normalized.extend( + _parse_requirement_group( + path, + project.get("dependencies"), + group_name="dependencies", + raw_type="project-dependency", + selection_kind="project", + ) + ) raw_optional = project.get("optional-dependencies", {}) if raw_optional is None: @@ -25,34 +41,58 @@ def parse(path: Path) -> list[Component]: optional_groups = require_mapping(raw_optional, f"{path}: project.optional-dependencies") for group_name, requirements in optional_groups.items(): if not isinstance(group_name, str): - raise ParseError(f"Malformed pyproject.toml in {path}: optional dependency group names must be strings.") + raise MalformedInputError( + f"Malformed pyproject.toml in {path}: optional dependency group names must be strings." + ) normalized.extend( _parse_requirement_group( path, requirements, - f"optional-dependencies.{group_name}", - "optional-dependency", + group_name=f"optional-dependencies.{group_name}", + raw_type="optional-dependency", + selection_kind="optional-dependency", ) ) return normalized +def _parse_dependency_group(path: Path, payload: dict[str, object], dependency_group: str) -> list[Component]: + dependency_groups, original_names = normalize_dependency_groups(payload.get("dependency-groups"), f"{path}: dependency-groups") + selected_group_name, resolved_requirements = resolve_dependency_group( + dependency_groups, + original_names, + requested_group=dependency_group, + context=str(path), + ) + return _parse_requirement_group( + path, + resolved_requirements, + group_name=f"dependency-groups.{selected_group_name}", + raw_type="dependency-group-dependency", + selection_kind="dependency-group", + ) + + def _parse_requirement_group( path: Path, raw_requirements: object, + *, group_name: str, raw_type: str, + selection_kind: str, ) -> list[Component]: if raw_requirements is None: return [] if not isinstance(raw_requirements, list): - raise ParseError(f"Malformed pyproject.toml in {path}: {group_name} must be an array of strings.") + raise MalformedInputError( + f"Malformed pyproject.toml in {path}: {group_name} must be an array of requirement strings." + ) normalized: list[Component] = [] for index, raw_requirement in enumerate(raw_requirements, start=1): if not isinstance(raw_requirement, str): - raise ParseError(f"Malformed pyproject.toml in {path}: {group_name}[{index}] must be a string.") + raise MalformedInputError(f"Malformed pyproject.toml in {path}: {group_name}[{index}] must be a string.") requirement = parse_requirement_text(raw_requirement, f"{path}:{group_name}[{index}]") version, exact_version = extract_requirement_version(requirement) normalized.append( @@ -69,6 +109,7 @@ def _parse_requirement_group( evidence={ "source_format": "pyproject-toml", "group": group_name, + "group_kind": selection_kind, "raw_requirement": raw_requirement, "specifier": str(requirement.specifier) or None, "marker": str(requirement.marker) if requirement.marker else None, diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/requirements_rules.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/requirements_rules.py new file mode 100644 index 0000000..b418759 --- /dev/null +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/requirements_rules.py @@ -0,0 +1,62 @@ +from __future__ import annotations + +import re +from pathlib import Path + +from ..errors import UnsupportedInputError + +_UNSUPPORTED_DIRECTIVE_PATTERNS: tuple[tuple[re.Pattern[str], str], ...] = ( + (re.compile(r"^(?:-r|--requirement)(?:\s|=|$)", re.IGNORECASE), "include directives (-r/--requirement)"), + (re.compile(r"^(?:-c|--constraint)(?:\s|=|$)", re.IGNORECASE), "constraint directives (-c/--constraint)"), + (re.compile(r"^(?:-e|--editable)(?:\s|=|$)", re.IGNORECASE), "editable installs (-e/--editable)"), + ( + re.compile(r"^(?:-i|--index-url|--extra-index-url|--no-index|--find-links|--trusted-host)(?:\s|=|$)", re.IGNORECASE), + "index and source options", + ), + ( + re.compile(r"^(?:--no-binary|--only-binary|--prefer-binary|--require-hashes|--hash|--pre|--all-releases|--only-final|--use-feature|--config-settings)(?:\s|=|$)", re.IGNORECASE), + "pip-specific options", + ), +) +_DIRECT_REFERENCE_MARKER_RE = re.compile(r"\s@\s") +_URL_PREFIX_RE = re.compile(r"^(?:https?|ftp|file|git\+|git|ssh)://", re.IGNORECASE) +_PLAIN_VCS_RE = re.compile(r"^(?:git|hg|svn|bzr)\+", re.IGNORECASE) +_LOCAL_PATH_RE = re.compile(r"^(?:\.{1,2}[\\/]|[\\/]|[A-Za-z]:[\\/])") +_ARCHIVE_SUFFIXES = ( + ".whl", + ".zip", + ".tar.gz", + ".tar.bz2", + ".tar.xz", + ".tgz", +) + + +def reject_unsupported_requirement_syntax(raw_requirement: str, *, path: Path, line_number: int) -> None: + stripped = raw_requirement.strip() + + for pattern, label in _UNSUPPORTED_DIRECTIVE_PATTERNS: + if pattern.match(stripped): + raise UnsupportedInputError( + f"Unsupported requirements.txt syntax in {path} at line {line_number}: {label} are not supported " + "in deterministic local mode." + ) + + if _DIRECT_REFERENCE_MARKER_RE.search(stripped): + raise UnsupportedInputError( + f"Unsupported requirements.txt syntax in {path} at line {line_number}: direct URL or path references " + "using '@' are not supported in deterministic local mode." + ) + + if _URL_PREFIX_RE.match(stripped) or _PLAIN_VCS_RE.match(stripped): + raise UnsupportedInputError( + f"Unsupported requirements.txt syntax in {path} at line {line_number}: archive URLs and VCS references " + "are not supported in deterministic local mode." + ) + + lowered = stripped.lower() + if _LOCAL_PATH_RE.match(stripped) or lowered.endswith(_ARCHIVE_SUFFIXES): + raise UnsupportedInputError( + f"Unsupported requirements.txt syntax in {path} at line {line_number}: local paths and archive paths " + "are not supported in deterministic local mode." + ) diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/requirements_txt.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/requirements_txt.py index b88c2ed..df399fe 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/requirements_txt.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/parsers/requirements_txt.py @@ -3,20 +3,23 @@ import re from pathlib import Path -from ..errors import ParseError +from ..errors import MalformedInputError, UnsupportedInputError from ..models import Component from .common import build_pypi_purl, extract_requirement_version, parse_requirement_text +from .requirements_rules import reject_unsupported_requirement_syntax def parse(path: Path) -> list[Component]: normalized: list[Component] = [] for start_line, raw_requirement in _iter_logical_requirements(path): - if raw_requirement.startswith("-"): - raise ParseError( - f"Unsupported requirements directive in {path} at line {start_line}: {raw_requirement!r}." - ) + reject_unsupported_requirement_syntax(raw_requirement, path=path, line_number=start_line) requirement = parse_requirement_text(raw_requirement, f"{path}:{start_line}") + if requirement.url is not None: + raise UnsupportedInputError( + f"Unsupported requirements.txt syntax in {path} at line {start_line}: " + "deterministic mode does not accept direct URL or VCS references." + ) version, exact_version = extract_requirement_version(requirement) normalized.append( Component( @@ -36,7 +39,7 @@ def parse(path: Path) -> list[Component]: "specifier": str(requirement.specifier) or None, "marker": str(requirement.marker) if requirement.marker else None, "extras": sorted(requirement.extras), - "url": requirement.url, + "url": None, }, ) ) @@ -75,7 +78,9 @@ def _iter_logical_requirements(path: Path) -> list[tuple[int, str]]: logical_lines.append((start_line, logical)) if buffer: - raise ParseError(f"Malformed requirements.txt in {path}: dangling line continuation starting at line {start_line}.") + raise MalformedInputError( + f"Malformed requirements.txt in {path}: dangling line continuation starting at line {start_line}." + ) return logical_lines diff --git a/tools/sbom-diff-and-risk/tests/fixtures/pyproject_groups_after.toml b/tools/sbom-diff-and-risk/tests/fixtures/pyproject_groups_after.toml new file mode 100644 index 0000000..e91aea6 --- /dev/null +++ b/tools/sbom-diff-and-risk/tests/fixtures/pyproject_groups_after.toml @@ -0,0 +1,25 @@ +[project] +name = "grouped-project" +version = "0.2.0" +dependencies = [ + "requests==2.32.0", +] + +[project.optional-dependencies] +docs = [ + "mkdocs>=1.6", +] + +[dependency-groups] +lint = [ + "ruff==0.6.3", +] +dev = [ + "pytest==8.3.1", + "mypy==1.11.2", + { include-group = "lint" }, +] +test = [ + "pytest-cov==5.0.0", + { include-group = "dev" }, +] diff --git a/tools/sbom-diff-and-risk/tests/fixtures/pyproject_groups_before.toml b/tools/sbom-diff-and-risk/tests/fixtures/pyproject_groups_before.toml new file mode 100644 index 0000000..67d38bd --- /dev/null +++ b/tools/sbom-diff-and-risk/tests/fixtures/pyproject_groups_before.toml @@ -0,0 +1,24 @@ +[project] +name = "grouped-project" +version = "0.1.0" +dependencies = [ + "requests==2.31.0", +] + +[project.optional-dependencies] +docs = [ + "mkdocs>=1.6", +] + +[dependency-groups] +lint = [ + "ruff==0.5.0", +] +dev = [ + "pytest==8.2.0", + { include-group = "lint" }, +] +test = [ + "pytest-cov==5.0.0", + { include-group = "dev" }, +] diff --git a/tools/sbom-diff-and-risk/tests/fixtures/pyproject_parser.toml b/tools/sbom-diff-and-risk/tests/fixtures/pyproject_parser.toml index 2b11c11..6f904c5 100644 --- a/tools/sbom-diff-and-risk/tests/fixtures/pyproject_parser.toml +++ b/tools/sbom-diff-and-risk/tests/fixtures/pyproject_parser.toml @@ -11,5 +11,5 @@ dev = [ "pytest>=8.0", ] docs = [ - "mkdocs @ https://example.com/packages/mkdocs-1.6.0-py3-none-any.whl", + "mkdocs>=1.6", ] diff --git a/tools/sbom-diff-and-risk/tests/fixtures/requirements_parser.txt b/tools/sbom-diff-and-risk/tests/fixtures/requirements_parser.txt index 0473cce..9ead1ac 100644 --- a/tools/sbom-diff-and-risk/tests/fixtures/requirements_parser.txt +++ b/tools/sbom-diff-and-risk/tests/fixtures/requirements_parser.txt @@ -1,8 +1,8 @@ # pinned requests==2.31.0 -# version range -urllib3>=2.0,<3.0 +# version range with marker +urllib3>=2.0,<3.0 ; python_version >= "3.11" -# direct reference -internal-lib @ https://example.com/packages/internal_lib-1.2.0-py3-none-any.whl ; python_version >= "3.11" +# extras remain local and deterministic +pytest[testing]>=8.0 diff --git a/tools/sbom-diff-and-risk/tests/test_cli_exit_codes.py b/tools/sbom-diff-and-risk/tests/test_cli_exit_codes.py index 12a3424..d184209 100644 --- a/tools/sbom-diff-and-risk/tests/test_cli_exit_codes.py +++ b/tools/sbom-diff-and-risk/tests/test_cli_exit_codes.py @@ -118,6 +118,7 @@ def test_cli_compare_help_mentions_policy_flags_and_exit_codes() -> None: assert result.returncode == 0 assert "--out-sarif" in result.stdout + assert "--pyproject-group" in result.stdout assert "--policy" in result.stdout assert "--fail-on" in result.stdout assert "--warn-on" in result.stdout @@ -148,6 +149,80 @@ def test_cli_can_write_sarif_only(tmp_path: Path) -> None: assert (tmp_path / "report.sarif").is_file() +def test_cli_pyproject_group_selection_smoke(tmp_path: Path) -> None: + project_root = Path(__file__).resolve().parents[1] + before = project_root / "examples" / "pyproject_groups_before.toml" + after = project_root / "examples" / "pyproject_groups_after.toml" + + result = _run_compare( + project_root, + [ + "--before", + str(before), + "--after", + str(after), + "--format", + "pyproject-toml", + "--pyproject-group", + "dev", + "--out-json", + str(tmp_path / "report.json"), + ], + ) + + assert result.returncode == 0 + assert (tmp_path / "report.json").is_file() + + +def test_cli_pyproject_group_missing_fails_clearly(tmp_path: Path) -> None: + project_root = Path(__file__).resolve().parents[1] + before = project_root / "examples" / "pyproject_groups_before.toml" + after = project_root / "examples" / "pyproject_groups_after.toml" + + result = _run_compare( + project_root, + [ + "--before", + str(before), + "--after", + str(after), + "--format", + "pyproject-toml", + "--pyproject-group", + "docs", + "--out-json", + str(tmp_path / "report.json"), + ], + ) + + assert result.returncode == 2 + assert "Requested dependency group" in result.stderr + assert "distinct from [project.optional-dependencies]" in result.stderr + + +def test_cli_pyproject_group_requires_pyproject_input(tmp_path: Path) -> None: + project_root = Path(__file__).resolve().parents[1] + before = project_root / "examples" / "requirements_before.txt" + after = project_root / "examples" / "requirements_after.txt" + + result = _run_compare( + project_root, + [ + "--before", + str(before), + "--after", + str(after), + "--pyproject-group", + "dev", + "--out-json", + str(tmp_path / "report.json"), + ], + ) + + assert result.returncode == 2 + assert "--pyproject-group requires at least one pyproject.toml input" in result.stderr + + def _run_compare(project_root: Path, args: list[str]) -> subprocess.CompletedProcess[str]: env = dict(os.environ) source_path = str(project_root / "src") diff --git a/tools/sbom-diff-and-risk/tests/test_normalize.py b/tools/sbom-diff-and-risk/tests/test_normalize.py index 52298a6..8cf3763 100644 --- a/tools/sbom-diff-and-risk/tests/test_normalize.py +++ b/tools/sbom-diff-and-risk/tests/test_normalize.py @@ -5,7 +5,7 @@ import pytest from sbom_diff_risk.errors import ParseError -from sbom_diff_risk.normalize import detect_format +from sbom_diff_risk.normalize import detect_format, normalize_input, normalize_input_with_options def test_detect_format_from_scaffold_fixtures() -> None: @@ -22,3 +22,29 @@ def test_detect_format_fails_clearly_for_malformed_json(tmp_path: Path) -> None: with pytest.raises(ParseError, match="Malformed JSON while detecting input format"): detect_format(broken) + + +def test_normalize_input_dispatches_to_parser() -> None: + fixture = Path(__file__).parent / "fixtures" / "requirements_before.txt" + + selected_format, components, notes = normalize_input(fixture) + + assert selected_format == "requirements-txt" + assert len(components) == 1 + assert components[0].name == "requests" + assert notes == [] + + +def test_normalize_input_with_pyproject_group_selects_dependency_group() -> None: + fixture = Path(__file__).parent / "fixtures" / "pyproject_groups_after.toml" + + selected_format, components, notes = normalize_input_with_options( + fixture, + declared_format="pyproject-toml", + pyproject_group="lint", + ) + + assert selected_format == "pyproject-toml" + assert [component.name for component in components] == ["ruff"] + assert components[0].version == "0.6.3" + assert notes == [] diff --git a/tools/sbom-diff-and-risk/tests/test_parsers.py b/tools/sbom-diff-and-risk/tests/test_parsers.py index bc64b99..8e4d870 100644 --- a/tools/sbom-diff-and-risk/tests/test_parsers.py +++ b/tools/sbom-diff-and-risk/tests/test_parsers.py @@ -4,8 +4,7 @@ import pytest -from sbom_diff_risk.errors import ParseError -from sbom_diff_risk.normalize import normalize_input +from sbom_diff_risk.errors import InputSelectionError, MalformedInputError, ParseError, UnsupportedInputError from sbom_diff_risk.parsers import cyclonedx_json, pyproject_toml, requirements_txt, spdx_json @@ -49,20 +48,35 @@ def test_spdx_parser_normalizes_component_fields() -> None: assert component.evidence["package"]["name"] == "requests" -def test_requirements_parser_normalizes_exact_range_and_direct_url() -> None: +def test_requirements_parser_normalizes_supported_pep508_subset() -> None: fixture = Path(__file__).parent / "fixtures" / "requirements_parser.txt" components = requirements_txt.parse(fixture) - assert [component.name for component in components] == ["requests", "urllib3", "internal-lib"] + assert [component.name for component in components] == ["requests", "urllib3", "pytest"] assert components[0].version == "2.31.0" assert components[0].purl == "pkg:pypi/requests@2.31.0" assert components[0].evidence["line_number"] == 2 assert components[1].version == "<3.0,>=2.0" - assert components[1].purl == "pkg:pypi/urllib3" - assert components[2].version is None - assert components[2].source_url == "https://example.com/packages/internal_lib-1.2.0-py3-none-any.whl" - assert components[2].evidence["marker"] == 'python_version >= "3.11"' + assert components[1].evidence["marker"] == 'python_version >= "3.11"' + assert components[2].version == ">=8.0" + assert components[2].evidence["extras"] == ["testing"] + assert components[2].source_url is None + + +def test_requirements_parser_supports_line_continuations(tmp_path: Path) -> None: + continuation_file = tmp_path / "requirements.txt" + continuation_file.write_text( + "urllib3>=2.0,\\\n<3.0 ; python_version >= \"3.11\"\n", + encoding="utf-8", + ) + + components = requirements_txt.parse(continuation_file) + + assert len(components) == 1 + assert components[0].name == "urllib3" + assert components[0].version == "<3.0,>=2.0" + assert components[0].evidence["marker"] == 'python_version >= "3.11"' def test_pyproject_parser_reads_project_and_optional_dependencies() -> None: @@ -73,28 +87,89 @@ def test_pyproject_parser_reads_project_and_optional_dependencies() -> None: assert [component.name for component in components] == ["requests", "urllib3", "pytest", "mkdocs"] assert components[0].raw_type == "project-dependency" assert components[2].evidence["group"] == "optional-dependencies.dev" - assert components[3].source_url == "https://example.com/packages/mkdocs-1.6.0-py3-none-any.whl" + assert components[2].evidence["group_kind"] == "optional-dependency" + assert components[3].source_url is None -def test_normalize_input_dispatches_to_parser() -> None: - fixture = Path(__file__).parent / "fixtures" / "requirements_before.txt" +def test_pyproject_parser_selects_dependency_group_with_includes() -> None: + fixture = Path(__file__).parent / "fixtures" / "pyproject_groups_before.toml" - selected_format, components, notes = normalize_input(fixture) + components = pyproject_toml.parse(fixture, dependency_group="test") + + assert [component.name for component in components] == ["pytest-cov", "pytest", "ruff"] + assert all(component.raw_type == "dependency-group-dependency" for component in components) + assert all(component.evidence["group_kind"] == "dependency-group" for component in components) + assert all(component.evidence["group"] == "dependency-groups.test" for component in components) - assert selected_format == "requirements-txt" - assert len(components) == 1 - assert components[0].name == "requests" - assert notes == [] +def test_pyproject_parser_normalizes_dependency_group_name_for_selection() -> None: + fixture = Path(__file__).parent / "fixtures" / "pyproject_groups_before.toml" -def test_requirements_parser_fails_clearly_on_unsupported_directive(tmp_path: Path) -> None: + components = pyproject_toml.parse(fixture, dependency_group="DEV") + + assert [component.name for component in components] == ["pytest", "ruff"] + + +@pytest.mark.parametrize( + ("line", "match"), + [ + ("-r base.txt\n", "include directives"), + ("-c constraints.txt\n", "constraint directives"), + ("-e .\n", "editable installs"), + ("package @ https://example.com/package.whl\n", "using '@'"), + ("https://example.com/package.whl\n", "archive URLs and VCS references"), + ("--index-url https://pypi.org/simple\n", "index and source options"), + ], +) +def test_requirements_parser_rejects_unsupported_deterministic_mode_syntax( + tmp_path: Path, + line: str, + match: str, +) -> None: broken = tmp_path / "requirements.txt" - broken.write_text("-r base.txt\n", encoding="utf-8") + broken.write_text(line, encoding="utf-8") - with pytest.raises(ParseError, match="Unsupported requirements directive"): + with pytest.raises(UnsupportedInputError, match=match): requirements_txt.parse(broken) +def test_requirements_parser_fails_on_malformed_continuation(tmp_path: Path) -> None: + malformed = tmp_path / "requirements.txt" + malformed.write_text("requests==2.31.0 \\\n", encoding="utf-8") + + with pytest.raises(MalformedInputError, match="dangling line continuation"): + requirements_txt.parse(malformed) + + +def test_pyproject_parser_requires_group_selection_for_dependency_groups_only_layout(tmp_path: Path) -> None: + broken = tmp_path / "pyproject.toml" + broken.write_text( + "[dependency-groups]\ndev = [\"pytest==8.2.0\"]\n", + encoding="utf-8", + ) + + with pytest.raises(InputSelectionError, match="select one explicitly with --pyproject-group"): + pyproject_toml.parse(broken) + + +def test_pyproject_parser_fails_when_requested_group_is_missing() -> None: + fixture = Path(__file__).parent / "fixtures" / "pyproject_groups_before.toml" + + with pytest.raises(InputSelectionError, match="distinct from \\[project.optional-dependencies\\]"): + pyproject_toml.parse(fixture, dependency_group="docs") + + +def test_pyproject_parser_fails_on_malformed_dependency_group_include(tmp_path: Path) -> None: + broken = tmp_path / "pyproject.toml" + broken.write_text( + "[dependency-groups]\ndev = [{ include-group = 42 }]\n", + encoding="utf-8", + ) + + with pytest.raises(MalformedInputError, match="include-group must be a non-empty string"): + pyproject_toml.parse(broken, dependency_group="dev") + + def test_pyproject_parser_fails_clearly_on_unsupported_layout(tmp_path: Path) -> None: broken = tmp_path / "pyproject.toml" broken.write_text("[tool.poetry]\nname = 'demo'\n", encoding="utf-8") From 2e6cbe16d31b4da94687a6a781c65e06e5e414cd Mon Sep 17 00:00:00 2001 From: stacknil Date: Sat, 18 Apr 2026 19:16:51 +0800 Subject: [PATCH 5/6] Polish sbom-diff-and-risk self provenance docs --- .github/workflows/sbom-diff-and-risk-ci.yml | 50 +++++++++ tools/sbom-diff-and-risk/README.md | 12 +++ .../docs/self-provenance.md | 100 ++++++++++++++++++ tools/sbom-diff-and-risk/pyproject.toml | 2 +- 4 files changed, 163 insertions(+), 1 deletion(-) create mode 100644 tools/sbom-diff-and-risk/docs/self-provenance.md diff --git a/.github/workflows/sbom-diff-and-risk-ci.yml b/.github/workflows/sbom-diff-and-risk-ci.yml index 301daf4..1c61320 100644 --- a/.github/workflows/sbom-diff-and-risk-ci.yml +++ b/.github/workflows/sbom-diff-and-risk-ci.yml @@ -11,9 +11,14 @@ on: - ".github/workflows/sbom-diff-and-risk-ci.yml" - "tools/sbom-diff-and-risk/**" +env: + SBOM_DIFF_RISK_DIST_ARTIFACT_NAME: sbom-diff-and-risk-dist + jobs: test: runs-on: ubuntu-latest + permissions: + contents: read defaults: run: working-directory: tools/sbom-diff-and-risk @@ -49,3 +54,48 @@ jobs: test -f "$tmpdir/report.md" diff -u examples/sample-report.json "$tmpdir/report.json" diff -u examples/sample-report.md "$tmpdir/report.md" + + build-and-attest: + # Keep provenance publication on trusted non-PR runs so consumers verify + # workflow-produced wheel and sdist artifacts from this repository workflow. + if: github.event_name != 'pull_request' + needs: test + runs-on: ubuntu-latest + permissions: + contents: read + id-token: write + attestations: write + defaults: + run: + working-directory: tools/sbom-diff-and-risk + steps: + - name: Check out repository + uses: actions/checkout@v4 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: "3.11" + + - name: Upgrade pip + run: python -m pip install --upgrade pip + + - name: Install build tooling + run: python -m pip install build + + - name: Build distributable artifacts + run: python -m build + + - name: Upload wheel and source distribution artifact + uses: actions/upload-artifact@v4 + with: + name: ${{ env.SBOM_DIFF_RISK_DIST_ARTIFACT_NAME }} + path: | + tools/sbom-diff-and-risk/dist/*.whl + tools/sbom-diff-and-risk/dist/*.tar.gz + if-no-files-found: error + + - name: Generate artifact attestation for built distributions + uses: actions/attest@v4 + with: + subject-path: ${{ github.workspace }}/tools/sbom-diff-and-risk/dist/* diff --git a/tools/sbom-diff-and-risk/README.md b/tools/sbom-diff-and-risk/README.md index 2abf653..4450aa1 100644 --- a/tools/sbom-diff-and-risk/README.md +++ b/tools/sbom-diff-and-risk/README.md @@ -226,6 +226,18 @@ sbom-diff-risk compare \ For GitHub code scanning integration guidance and a minimal upload workflow, see [docs/github-code-scanning.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/docs/github-code-scanning.md). +## Self-provenance + +This repository also records provenance for `sbom-diff-and-risk` itself by generating GitHub artifact attestations for the wheel and source distribution produced by the `sbom-diff-and-risk-ci` workflow. + +- the attested files are the wheel and source distribution built by `python -m build` from `tools/sbom-diff-and-risk` +- the build files are uploaded together as the `sbom-diff-and-risk-dist` workflow artifact +- only trusted non-PR runs publish the attestation +- consumers can verify provenance with GitHub's attestation tooling after downloading one of those artifacts +- this complements the tool's analysis of third-party supply-chain inputs, but it does not replace that analysis + +See [docs/self-provenance.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/docs/self-provenance.md) for the exact attested filenames, where the evidence appears in GitHub, and a run-by-run verification flow for consumers. + ## Parser Boundaries Deterministic local mode intentionally supports a conservative subset of packaging syntax. The detailed matrix lives in [docs/parser-boundaries.md](D:/OneDrive/Code/scientific-computing-toolkit/tools/sbom-diff-and-risk/docs/parser-boundaries.md). diff --git a/tools/sbom-diff-and-risk/docs/self-provenance.md b/tools/sbom-diff-and-risk/docs/self-provenance.md new file mode 100644 index 0000000..96db8ee --- /dev/null +++ b/tools/sbom-diff-and-risk/docs/self-provenance.md @@ -0,0 +1,100 @@ +# Self-provenance and artifact attestations + +`sbom-diff-and-risk` analyzes third-party dependency changes, but consumers should also be able to verify where the tool itself came from. This repository generates GitHub artifact attestations for the packaged build outputs produced by the `sbom-diff-and-risk-ci` workflow. + +## What is attested in this repository + +The attested subjects are the exact Python distributables built from `tools/sbom-diff-and-risk` via `python -m build`: + +- the wheel: `dist/sbom_diff_and_risk--py3-none-any.whl` +- the source distribution: `dist/sbom_diff_and_risk-.tar.gz` + +Those two files are uploaded together as the workflow artifact named `sbom-diff-and-risk-dist`. The attestation applies to the built files themselves, not just to the artifact bundle name shown in the Actions UI. + +Current attestations cover workflow-built wheel and sdist artifacts, not GitHub Release assets or PyPI-published distributions. + +## Workflow and permissions + +The attestation is generated in `.github/workflows/sbom-diff-and-risk-ci.yml` by the `build-and-attest` job in the `sbom-diff-and-risk-ci` workflow. + +That job runs only for trusted non-PR events in this repository: + +- `push` +- `workflow_dispatch` + +Pull request runs still execute the `test` job, but they do not publish artifact attestations. + +The `build-and-attest` job uses the minimum explicit permissions required for GitHub-hosted build provenance: + +- `contents: read` for repository checkout +- `id-token: write` for GitHub's signing identity +- `attestations: write` to publish the attestation + +## Where provenance evidence appears in GitHub + +After a successful non-PR run of `sbom-diff-and-risk-ci`, consumers can find the evidence in two useful places: + +1. On the workflow run page: + - the uploaded artifact appears as `sbom-diff-and-risk-dist` + - this is the run consumers should use to confirm the workflow name, job name, and downloaded artifact bundle before verification +2. In the repository-wide attestations view: + - open **Actions** + - in the left sidebar, under **Management**, open **Attestations** + - search for `sbom_diff_and_risk-` or filter by recent creation date + +On the **Attestations** page, the relevant subjects are the wheel and sdist filenames, not the workflow artifact bundle name. On the workflow run page, the main visible bundle name is still `sbom-diff-and-risk-dist`. + +## Manual verification for one workflow run + +Use this path after a merge to the default branch or an intentional `workflow_dispatch` run. + +1. Open the repository's **Actions** tab. +2. Open a successful `sbom-diff-and-risk-ci` run triggered by `push` or `workflow_dispatch`. +3. Confirm that the `build-and-attest` job ran successfully. +4. Download the `sbom-diff-and-risk-dist` artifact from that run. +5. Confirm the downloaded archive contains exactly the expected build outputs for that version: + - `sbom_diff_and_risk--py3-none-any.whl` + - `sbom_diff_and_risk-.tar.gz` +6. Verify one of the files with the GitHub CLI: + +```bash +gh attestation verify path/to/sbom_diff_and_risk--py3-none-any.whl \ + --repo OWNER/scientific-computing-toolkit \ + --signer-workflow OWNER/scientific-computing-toolkit/.github/workflows/sbom-diff-and-risk-ci.yml +``` + +You can verify the source distribution the same way: + +```bash +gh attestation verify path/to/sbom_diff_and_risk-.tar.gz \ + --repo OWNER/scientific-computing-toolkit \ + --signer-workflow OWNER/scientific-computing-toolkit/.github/workflows/sbom-diff-and-risk-ci.yml +``` + +If you want more inspection detail during review, ask the CLI for structured output: + +```bash +gh attestation verify path/to/sbom_diff_and_risk--py3-none-any.whl \ + --repo OWNER/scientific-computing-toolkit \ + --signer-workflow OWNER/scientific-computing-toolkit/.github/workflows/sbom-diff-and-risk-ci.yml \ + --format json +``` + +A successful verification confirms that: + +- the downloaded file matches an attested subject +- the attestation was linked to `OWNER/scientific-computing-toolkit` +- the attestation was signed by `.github/workflows/sbom-diff-and-risk-ci.yml` + +## Release-consumer note + +If these same wheel or source distribution bytes are later attached to a GitHub release, consumers should verify the downloaded release asset file itself with the same `gh attestation verify` flow. In the current setup, the provenance source of truth is still the workflow-produced build artifact and its attestation, not a separate release-attestation workflow. + +## How this complements the tool's own analysis + +Self-provenance and dependency analysis solve different problems: + +- artifact attestations help consumers verify where `sbom-diff-and-risk` itself was built +- `sbom-diff-and-risk` helps users review and gate third-party dependency changes in their own projects + +These attestations strengthen trust in the tool's own distributable artifacts, but they do not replace the tool's analysis of external SBOM inputs, policy decisions, or trust-signal reporting for third-party packages. diff --git a/tools/sbom-diff-and-risk/pyproject.toml b/tools/sbom-diff-and-risk/pyproject.toml index 1a3e167..3e7cedb 100644 --- a/tools/sbom-diff-and-risk/pyproject.toml +++ b/tools/sbom-diff-and-risk/pyproject.toml @@ -8,7 +8,7 @@ version = "0.1.0" description = "Local, deterministic SBOM diff and heuristic risk reporting." readme = "README.md" requires-python = ">=3.11" -license = { text = "MIT" } +license = "MIT" authors = [ { name = "OpenAI Codex" } ] From 4b05fdbc02fafa1ac409c601d9eb904a6111c541 Mon Sep 17 00:00:00 2001 From: stacknil Date: Sat, 18 Apr 2026 19:19:21 +0800 Subject: [PATCH 6/6] Normalize policy report paths across platforms --- .../examples/sample-policy-fail-report.json | 4 ++-- .../examples/sample-policy-fail-report.md | 2 +- .../examples/sample-policy-warn-report.json | 4 ++-- .../examples/sample-policy-warn-report.md | 2 +- .../src/sbom_diff_risk/policy_parser.py | 10 +++++++++- 5 files changed, 15 insertions(+), 7 deletions(-) diff --git a/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.json b/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.json index a5b7271..3469a60 100644 --- a/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.json +++ b/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.json @@ -302,7 +302,7 @@ ], "policy_evaluation": { "applied": true, - "policy_path": "examples\\policy-strict.yml", + "policy_path": "examples/policy-strict.yml", "effective_policy": { "version": 1, "block_on": [ @@ -485,7 +485,7 @@ "stub": false, "policy_evaluation": { "applied": true, - "policy_path": "examples\\policy-strict.yml", + "policy_path": "examples/policy-strict.yml", "effective_policy": { "version": 1, "block_on": [ diff --git a/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.md b/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.md index 20485a7..f725fc7 100644 --- a/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.md +++ b/tools/sbom-diff-and-risk/examples/sample-policy-fail-report.md @@ -18,7 +18,7 @@ ## Policy summary - Applied: yes -- Policy path: examples\policy-strict.yml +- Policy path: examples/policy-strict.yml - Exit code: 1 - Blocking findings: 3 - Warnings: 1 diff --git a/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.json b/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.json index 4410c59..ba0d7c7 100644 --- a/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.json +++ b/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.json @@ -302,7 +302,7 @@ ], "policy_evaluation": { "applied": true, - "policy_path": "examples\\policy-minimal.yml", + "policy_path": "examples/policy-minimal.yml", "effective_policy": { "version": 1, "block_on": [ @@ -420,7 +420,7 @@ "stub": false, "policy_evaluation": { "applied": true, - "policy_path": "examples\\policy-minimal.yml", + "policy_path": "examples/policy-minimal.yml", "effective_policy": { "version": 1, "block_on": [ diff --git a/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.md b/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.md index 9422e2b..d4ce8d1 100644 --- a/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.md +++ b/tools/sbom-diff-and-risk/examples/sample-policy-warn-report.md @@ -18,7 +18,7 @@ ## Policy summary - Applied: yes -- Policy path: examples\policy-minimal.yml +- Policy path: examples/policy-minimal.yml - Exit code: 0 - Blocking findings: 0 - Warnings: 1 diff --git a/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_parser.py b/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_parser.py index 6b417bf..eb11b5c 100644 --- a/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_parser.py +++ b/tools/sbom-diff-and-risk/src/sbom_diff_risk/policy_parser.py @@ -72,7 +72,7 @@ def build_policy( rendered_path: str | None = None if policy_path is not None: base_policy = load_policy(policy_path) - rendered_path = str(policy_path) + rendered_path = _render_policy_path(policy_path) cli_block_on = parse_rule_csv(fail_on, "--fail-on") cli_warn_on = parse_rule_csv(warn_on, "--warn-on") @@ -154,3 +154,11 @@ def _validate_rule_ids(rule_ids: Iterable[str], context: str) -> tuple[str, ...] def _merge_strings(base: tuple[str, ...], extra: tuple[str, ...]) -> tuple[str, ...]: return tuple(dict.fromkeys((*base, *extra))) + + +def _render_policy_path(policy_path: Path) -> str: + resolved_policy_path = policy_path.resolve() + try: + return resolved_policy_path.relative_to(Path.cwd().resolve()).as_posix() + except ValueError: + return resolved_policy_path.as_posix()