diff --git a/ROADMAP.md b/ROADMAP.md index e603b1a..63fa6e0 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -3,12 +3,12 @@ schema: aether.architecture-document/v1 id: renderflow-roadmap title: Renderflow Roadmap kind: architecture-document -version: 0.1.2 +version: 0.1.3 status: draft owners: - egohygiene created: 2026-08-19 -updated: 2026-09-25 +updated: 2026-09-27 governed_by: - architecture-roadmap depends_on: @@ -26,6 +26,21 @@ supersedes: [] # Renderflow Roadmap +## 2026-09-27 ordered collection review handoff + +Checkpoint #415 adds canonical planning and execution for one explicitly +ordered, homogeneous local artifact collection. Members require stable IDs, +root-relative locators, declared formats/media types, and SHA-256 digests; +optional geometry enters plan identity. Planning freezes member evidence, +execution rechecks paths and bytes, and run evidence records source order and +aggregate lineage. The two-page synthetic fixture and generated 44-member +recipe prove the public SDK and CLI paths without claiming PDF or EPUB output. + +After review and merge, #416 (print PDF) and #417 (fixed-layout EPUB) may +proceed independently. #418 validates the EPUB capability, and #419 publishes +the immutable release consumed by Flow #52. No real publication artifacts or +personal media were processed by the #415 tests. + ## 2026-09-25 live suite handoff > [!IMPORTANT] diff --git a/crates/renderflow-cli/Cargo.toml b/crates/renderflow-cli/Cargo.toml index 04d2d1a..392ea85 100644 --- a/crates/renderflow-cli/Cargo.toml +++ b/crates/renderflow-cli/Cargo.toml @@ -27,6 +27,10 @@ serde_json = "1" name = "cli_tests" path = "../../tests/cli_tests.rs" +[[test]] +name = "ordered_collection_cli" +path = "../../tests/ordered_collection_cli.rs" + [[test]] name = "graph_integration_test" path = "../../tests/graph_integration_test.rs" diff --git a/crates/renderflow-core/benches/cache.rs b/crates/renderflow-core/benches/cache.rs index d1c1007..61656f2 100644 --- a/crates/renderflow-core/benches/cache.rs +++ b/crates/renderflow-core/benches/cache.rs @@ -116,7 +116,7 @@ fn bench_transform_cache_lookup(c: &mut Criterion) { fn bench_transform_cache_insert(c: &mut Criterion) { c.bench_function("transform_cache/insert", |b| { b.iter_batched( - || TransformCache::default(), + TransformCache::default, |mut cache| { cache.insert("abc123".to_string(), "transformed output".to_string()); cache diff --git a/crates/renderflow-core/src/graph/dag_executor.rs b/crates/renderflow-core/src/graph/dag_executor.rs index bcf1eb5..118afd0 100644 --- a/crates/renderflow-core/src/graph/dag_executor.rs +++ b/crates/renderflow-core/src/graph/dag_executor.rs @@ -521,6 +521,36 @@ impl DagExecutor { .into_result() } + /// Execute an ordered collection and retain step evidence for canonical runs. + /// The root collection stays in the caller's source evidence; every derived + /// format must resolve to one artifact for the current output lifecycle. + pub fn execute_collection_with_evidence( + &self, + dag: &MultiTargetDag, + source_format: Format, + initial_artifacts: ArtifactCollection, + store: &ArtifactStore, + ) -> Result { + let report = + self.execute_artifacts_with_evidence(dag, source_format, initial_artifacts, store)?; + let artifacts = report + .artifacts + .into_iter() + .filter(|(format, _)| *format != source_format) + .map(|(format, collection)| { + let artifact = collection.into_one().with_context(|| { + format!("Format '{format}' produced multiple artifacts where one was expected") + })?; + Ok((format, artifact)) + }) + .collect::>>()?; + Ok(DagExecutionReport { + artifacts, + steps: report.steps, + diagnostics: report.diagnostics, + }) + } + fn execute_artifacts_with_evidence( &self, dag: &MultiTargetDag, @@ -1111,7 +1141,7 @@ impl DagExecutor { output, transform.name().to_string(), "unstable-v1".to_string(), - sha256_text(transform.name()), + edge_configuration_digest(edge), None, )) } diff --git a/crates/renderflow-core/src/graph/execution_plan.rs b/crates/renderflow-core/src/graph/execution_plan.rs index 0b09e36..f55e664 100644 --- a/crates/renderflow-core/src/graph/execution_plan.rs +++ b/crates/renderflow-core/src/graph/execution_plan.rs @@ -154,6 +154,24 @@ pub struct PlanSourceArtifact { pub profile: ResolvedArtifactProfile, } +/// Ordered, explicitly declared collection identity frozen at planning time. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct PlanSourceCollection { + pub source_id: String, + pub members: Vec, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct PlanCollectionMember { + pub source_id: String, + pub locator: String, + pub media_type: String, + pub format: String, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub geometry: Option, + pub artifact: PlanSourceArtifact, +} + impl From<&IntakeReport> for PlanSourceArtifact { fn from(report: &IntakeReport) -> Self { Self { @@ -288,6 +306,9 @@ pub struct ExecutionPlan { /// Source identity and multi-signal evidence established before planning. #[serde(default, skip_serializing_if = "Option::is_none")] pub source_artifact: Option, + /// Present instead of `source_artifact` for a declared ordered collection. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub source_collection: Option, /// Reproducible evidence for providers selected by this exact plan. #[serde(default, skip_serializing_if = "Option::is_none")] pub toolchain: Option, @@ -405,6 +426,7 @@ impl ExecutionPlan { metadata, diagnostics, source_artifact: None, + source_collection: None, toolchain: None, artifact_forest: None, } diff --git a/crates/renderflow-core/src/graph/mod.rs b/crates/renderflow-core/src/graph/mod.rs index 2572a0c..7e522d1 100644 --- a/crates/renderflow-core/src/graph/mod.rs +++ b/crates/renderflow-core/src/graph/mod.rs @@ -15,7 +15,7 @@ pub use definition::TransformDefinition; pub use definition_registry::TransformDefinitionRegistry; pub use execution_plan::{ ArtifactForest, DiagnosticLevel, ExecutionPlan, ForestBranch, ForestBranchState, - PlanSourceArtifact, + PlanCollectionMember, PlanSourceArtifact, PlanSourceCollection, }; pub use format::Format; pub use input_kind::InputKind; diff --git a/crates/renderflow-core/src/planning.rs b/crates/renderflow-core/src/planning.rs index b63ce7f..d60f37e 100644 --- a/crates/renderflow-core/src/planning.rs +++ b/crates/renderflow-core/src/planning.rs @@ -15,7 +15,9 @@ use anyhow::{Context, Result}; use crate::adapters::strategy::{ document_input_format, output_type_for_format, StrategyArtifactTransform, }; -use crate::artifact::{Artifact, ArtifactDescriptor, ArtifactStorageClass, ArtifactStore}; +use crate::artifact::{ + Artifact, ArtifactCollection, ArtifactDescriptor, ArtifactStorageClass, ArtifactStore, +}; use crate::checkpoint::CheckpointContext; use crate::evidence::{ redact_sensitive_text, run_id, sha256_serialized, unix_time_ms, ArtifactEvidence, @@ -26,10 +28,11 @@ use crate::evidence::{ use crate::graph::capability::{FormatCapabilityRegistry, FormatFamily}; use crate::graph::{ ArtifactForest, DagExecutionReport, DagExecutor, DiagnosticLevel, ExecutionPlan, ForestBranch, - ForestBranchState, Format, MultiTargetDag, TransformEdge, TransformGraph, + ForestBranchState, Format, MultiTargetDag, PlanCollectionMember, PlanSourceArtifact, + PlanSourceCollection, TransformEdge, TransformGraph, }; use crate::hygiene::{HygieneEngine, HygieneEvidence}; -use crate::intake::{IntakeEngine, IntakeRequest, ResolvedArtifactProfile}; +use crate::intake::{IntakeEngine, IntakeReport, IntakeRequest, ResolvedArtifactProfile}; use crate::optimization::OptimizationMode; use crate::publication::write_release_metadata; use crate::spec::{ @@ -156,12 +159,13 @@ impl ResolvedTarget { pub struct ResolvedExecution { plan: ExecutionPlan, + config_path: PathBuf, spec: SpecV2, source_version: SourceSpecVersion, source: SourceSpec, source_path: PathBuf, source_format: Format, - source_profile: ResolvedArtifactProfile, + source_members: Vec, targets: Vec, dag: MultiTargetDag, executor: DagExecutor, @@ -170,6 +174,13 @@ pub struct ResolvedExecution { cancellation: Option>, } +#[derive(Debug, Clone)] +struct ResolvedSourceMember { + spec: SourceSpec, + path: PathBuf, + intake: IntakeReport, +} + impl ResolvedExecution { pub fn plan(&self) -> &ExecutionPlan { &self.plan @@ -282,21 +293,28 @@ pub fn resolve(request: PlanningRequest) -> Result { compose_selected_profiles(&mut spec)?; let source = select_primary_source(&spec)?; - let source_path = resolve_path_relative_to_config( - &request.config_path, - source.path.as_deref().ok_or_else(|| { - anyhow::anyhow!( - "canonical execution currently requires a local source path for '{}'", - source.id - ) - })?, - ); - if !source_path.is_file() { - anyhow::bail!( - "source artifact '{}' does not exist or is not a file", - source_path.display() - ); - } + let collection = source.kind == SourceKind::Collection; + let config_path = if collection { + fs::canonicalize(&request.config_path) + .context("collection.source.config: unable to resolve spec path")? + } else { + request.config_path.clone() + }; + let member_specs = if collection { + source + .members + .iter() + .map(|id| { + spec.sources + .iter() + .find(|item| &item.id == id) + .cloned() + .with_context(|| format!("collection.member.unknown: '{id}'")) + }) + .collect::>>()? + } else { + vec![source.clone()] + }; if let Some(publication) = &spec.publication { for artwork in &publication.artwork { let path = resolve_path_relative_to_config(&request.config_path, &artwork.path); @@ -315,17 +333,81 @@ pub fn resolve(request: PlanningRequest) -> Result { // run store and verifies the digest through normal artifact handling. let intake_staging = tempfile::tempdir().context("failed to create intake staging store")?; let intake_store = ArtifactStore::new(intake_staging.path())?; - let mut intake_request = IntakeRequest::from_path(&source_path); - if let Some(format) = &source.format { - intake_request = intake_request.with_format(format.clone()); - } - if let Some(media_type) = &source.media_type { - intake_request = intake_request.with_media_type(media_type.clone()); + let mut source_members = Vec::new(); + let mut source_format = None; + for member in member_specs { + let path = if collection { + collection_member_path(&config_path, &member)? + } else { + let path = resolve_path_relative_to_config( + &request.config_path, + member.path.as_deref().ok_or_else(|| { + anyhow::anyhow!( + "canonical execution requires a local source path for '{}'", + member.id + ) + })?, + ); + if !path.is_file() { + anyhow::bail!( + "source artifact '{}' does not exist or is not a file", + path.display() + ); + } + path + }; + let intake = intake_source(&member, &path, &intake_store)?; + let format = resolve_source_format(&member, &path, &intake.profile)?; + if collection { + let declared_media = member + .media_type + .as_deref() + .context("collection.member.media: missing declared media type")?; + let supported = FormatCapabilityRegistry::global() + .get(format) + .is_some_and(|descriptor| descriptor.media_types.contains(&declared_media)); + if !supported { + anyhow::bail!( + "collection.member.unsupported_media: '{}' declares '{}' for format '{}'", + member.id, + declared_media, + format + ); + } + if !intake.profile.conflicts.is_empty() { + anyhow::bail!( + "collection.member.format_conflict: '{}' has conflicting format signals", + member.id + ); + } + if member.sha256.as_deref() != Some(intake.source.digest().value()) { + anyhow::bail!( + "collection.member.digest_mismatch: '{}' differs from declared sha256", + member.id + ); + } + if member.media_type.as_deref() != Some(intake.profile.media_type.as_str()) { + anyhow::bail!( + "collection.member.media_mismatch: '{}' has media type '{}', expected {:?}", + member.id, + intake.profile.media_type, + member.media_type + ); + } + if source_format.is_some_and(|selected| selected != format) { + anyhow::bail!("collection.member.unsupported_media: '{}' has format '{}'; collection members must share one format", member.id, format); + } + } + source_format = Some(format); + source_members.push(ResolvedSourceMember { + spec: member, + path, + intake, + }); } - let source_intake = IntakeEngine::new() - .intake(&intake_request, &intake_store) - .context("failed to establish source artifact identity before planning")?; - let source_format = resolve_source_format(&source, &source_path, &source_intake.profile)?; + let source_format = source_format.context("source collection has no members")?; + let source_path = source_members[0].path.clone(); + let source_intake = &source_members[0].intake; let (mut graph, mut executor, mut tool_registry) = if let Some(transforms_path) = &spec.transforms { @@ -350,6 +432,12 @@ pub fn resolve(request: PlanningRequest) -> Result { register_builtin_strategy_edges(&mut graph, &mut tool_registry)?; let policy_graph = apply_execution_policy(&graph, &tool_registry, &spec); + let policy_graph = if collection { + policy_graph + .filtered_by(|edge| edge.from != source_format || edge.input_kind.is_collection()) + } else { + policy_graph + }; let requested_targets = resolve_target_intent(&spec, &policy_graph, source_format)?; if requested_targets.is_empty() { anyhow::bail!("target selection resolved to no executable artifact formats"); @@ -398,6 +486,16 @@ pub fn resolve(request: PlanningRequest) -> Result { (dag, true) } }; + if collection + && (dag.all_edges().is_empty() + || dag + .all_edges() + .iter() + .any(|edge| edge.from == source_format && !edge.input_kind.is_collection()) + || targets.iter().any(|target| target.format == source_format)) + { + anyhow::bail!("collection.transform.unsupported: a collection requires a registered collection-input transform and a distinct output format"); + } register_builtin_strategy_executors( &mut executor, @@ -418,7 +516,24 @@ pub fn resolve(request: PlanningRequest) -> Result { &pruned_unavailable, &pruned_budget, )); - plan.attach_source_artifact(&source_intake); + if collection { + plan.source_collection = Some(PlanSourceCollection { + source_id: source.id.clone(), + members: source_members + .iter() + .map(|member| PlanCollectionMember { + source_id: member.spec.id.clone(), + locator: member.spec.path.clone().unwrap_or_default(), + media_type: member.intake.profile.media_type.clone(), + format: source_format.to_string(), + geometry: member.spec.geometry.clone(), + artifact: PlanSourceArtifact::from(&member.intake), + }) + .collect(), + }); + } else { + plan.attach_source_artifact(source_intake); + } if !source_intake.profile.conflicts.is_empty() { plan.add_tool_diagnostic(format!( "source intake reported {} conflicting format signal set(s); '{}' was selected", @@ -483,12 +598,13 @@ pub fn resolve(request: PlanningRequest) -> Result { Ok(ResolvedExecution { plan, + config_path, spec, source_version, source, source_path, source_format, - source_profile: source_intake.profile, + source_members, targets, dag, executor, @@ -617,47 +733,74 @@ pub fn execute(mut resolved: ResolvedExecution, dry_run: bool) -> Result Result { + if resolved.source.kind == SourceKind::Collection { + collection_member_path(&resolved.config_path, &member.spec)?; + } + let descriptor = ArtifactDescriptor::for_format( + resolved.source_format, + ArtifactStorageClass::Source, + ) + .with_metadata("renderflow.source_id", member.spec.id.clone()) .with_metadata("renderflow.intake.schema", crate::intake::INTAKE_SCHEMA_V1) .with_metadata( "renderflow.intake.profile", - serde_json::to_value(&resolved.source_profile)?, - ), - ) { - Ok(artifact) => artifact, - Err(error) => { - return failed_execution_result( - &resolved, - &output_root, - started_at_unix_ms, - Vec::new(), - Vec::new(), - "execution.source_import_failed", - error, - ) + serde_json::to_value(&member.intake.profile)?, + ); + let descriptor = if resolved.source.kind == SourceKind::Collection { + descriptor + .with_metadata("renderflow.collection.id", resolved.source.id.clone()) + .with_metadata("renderflow.collection.index", index) + .with_metadata("renderflow.collection.locator", member.spec.path.clone()) + .with_metadata( + "renderflow.collection.geometry", + serde_json::to_value(&member.spec.geometry)?, + ) + } else { + descriptor + }; + let artifact = store + .import_path(&member.path, descriptor) + .with_context(|| format!("collection.member.unreadable: '{}'", member.spec.id))?; + if resolved.source.kind == SourceKind::Collection { + collection_member_path(&resolved.config_path, &member.spec)?; + } + let planned_digest = if let Some(collection) = &resolved.plan.source_collection { + &collection.members[index].artifact.digest + } else { + &resolved + .plan + .source_artifact + .as_ref() + .context("missing source artifact plan")? + .digest + }; + if artifact.digest().to_string() != *planned_digest { + anyhow::bail!( + "collection.member.changed: '{}' differs from planned bytes; re-plan", + member.spec.id + ); + } + Ok(artifact) + })(); + match imported { + Ok(artifact) => source_artifacts.push(artifact), + Err(error) => { + return failed_execution_result( + &resolved, + &output_root, + started_at_unix_ms, + source_evidence(&resolved, &source_artifacts), + Vec::new(), + "execution.source_changed_after_intake", + error, + ) + } } - }; - if resolved - .plan - .source_artifact - .as_ref() - .is_some_and(|planned| planned.digest != source_artifact.digest().to_string()) - { - return failed_execution_result( - &resolved, - &output_root, - started_at_unix_ms, - vec![source_artifact_evidence(&resolved, &source_artifact)], - Vec::new(), - "execution.source_changed_after_intake", - anyhow::anyhow!( - "source bytes changed after intake and before transform execution; re-plan the run" - ), - ); } + let source_artifact = source_artifacts[0].clone(); let executor = std::mem::take(&mut resolved.executor); let checkpoint_context = CheckpointContext { @@ -669,8 +812,16 @@ pub fn execute(mut resolved: ResolvedExecution, dry_run: bool) -> Result Result report, Err(error) => { - let source_evidence = source_artifact_evidence(&resolved, &source_artifact); + let source_evidence = source_evidence(&resolved, &source_artifacts); return failed_execution_result( &resolved, &output_root, started_at_unix_ms, - vec![source_evidence], + source_evidence, Vec::new(), "execution.executor_failed", error, @@ -932,7 +1093,7 @@ pub fn execute(mut resolved: ResolvedExecution, dry_run: bool) -> Result Vec { .collect() } -fn source_artifact_evidence(resolved: &ResolvedExecution, source: &Artifact) -> ArtifactEvidence { - ArtifactEvidence::from_artifact( - source, - resolved - .source - .role - .clone() - .unwrap_or_else(|| resolved.source.id.clone()), - ArtifactRole::Source, - artifact_store_locator(source), - ProducerEvidence::source(), - ValidationState::NotRequested, - FidelityDeclaration::Lossless, - ) +fn source_evidence(resolved: &ResolvedExecution, sources: &[Artifact]) -> Vec { + resolved + .source_members + .iter() + .zip(sources) + .map(|(member, source)| { + ArtifactEvidence::from_artifact( + source, + member + .spec + .role + .clone() + .unwrap_or_else(|| member.spec.id.clone()), + ArtifactRole::Source, + artifact_store_locator(source), + ProducerEvidence::source(), + ValidationState::NotRequested, + FidelityDeclaration::Lossless, + ) + }) + .collect() } fn artifact_evidence( resolved: &ResolvedExecution, - source: &Artifact, + sources: &[Artifact], report: &DagExecutionReport, output_locators: &HashMap, diagnostics: &[ExecutionDiagnostic], validation_outcomes: &HashMap, hygiene_evidence: &HashMap, ) -> Vec { - let mut evidence = vec![source_artifact_evidence(resolved, source)]; + let mut evidence = source_evidence(resolved, sources); let mut artifacts = report.artifacts.iter().collect::>(); artifacts.sort_by(|(left_format, left), (right_format, right)| { left_format @@ -1244,7 +1412,7 @@ fn artifact_evidence( .then_with(|| left.id().as_str().cmp(right.id().as_str())) }); for (format, artifact) in artifacts { - if artifact.id() == source.id() + if sources.iter().any(|source| artifact.id() == source.id()) && !resolved .targets .iter() @@ -1642,6 +1810,31 @@ fn merge_selector_set(destination: &mut SelectorSet, source: &SelectorSet) { } fn select_primary_source(spec: &SpecV2) -> Result { + let collections: Vec<&SourceSpec> = spec + .sources + .iter() + .filter(|source| source.kind == SourceKind::Collection) + .collect(); + if let [collection] = collections.as_slice() { + let declared = spec + .sources + .iter() + .filter(|source| source.kind == SourceKind::Artifact) + .map(|source| source.id.as_str()) + .collect::>(); + let selected = collection + .members + .iter() + .map(String::as_str) + .collect::>(); + if declared != selected { + anyhow::bail!("collection.source.ambiguous: every declared artifact must occur exactly once in the selected collection"); + } + return Ok((*collection).clone()); + } + if collections.len() > 1 { + anyhow::bail!("collection.source.ambiguous: canonical execution requires exactly one collection source"); + } let artifacts: Vec<&SourceSpec> = spec .sources .iter() @@ -1649,13 +1842,73 @@ fn select_primary_source(spec: &SpecV2) -> Result { .collect(); if artifacts.len() != 1 { anyhow::bail!( - "canonical format-DAG execution currently requires exactly one artifact source; found {}. Multi-root/collection source identity is reserved for the Transform v2 execution graph (#357).", + "canonical execution requires one artifact source or one explicitly ordered collection; found {} artifacts", artifacts.len() ); } Ok(artifacts[0].clone()) } +fn collection_member_path(config_path: &Path, source: &SourceSpec) -> Result { + let locator = source.path.as_deref().ok_or_else(|| { + anyhow::anyhow!("collection.member.locator: '{}' requires a path", source.id) + })?; + let relative = Path::new(locator); + if relative.as_os_str().is_empty() + || !relative + .components() + .all(|component| matches!(component, Component::Normal(_))) + { + anyhow::bail!( + "collection.member.locator: '{}' must use a root-relative path without traversal", + source.id + ); + } + let root = config_path + .parent() + .filter(|parent| !parent.as_os_str().is_empty()) + .unwrap_or_else(|| Path::new(".")); + let mut path = root.to_path_buf(); + for component in relative.components() { + path.push(component); + let metadata = fs::symlink_metadata(&path).with_context(|| { + format!( + "collection.member.missing: '{}' at '{}'", + source.id, + path.display() + ) + })?; + if metadata.file_type().is_symlink() { + anyhow::bail!( + "collection.member.symlink: '{}' has a symlink component at '{}'", + source.id, + path.display() + ); + } + } + if !path.is_file() { + anyhow::bail!( + "collection.member.not_file: '{}' at '{}'", + source.id, + path.display() + ); + } + Ok(path) +} + +fn intake_source(source: &SourceSpec, path: &Path, store: &ArtifactStore) -> Result { + let mut request = IntakeRequest::from_path(path); + if let Some(format) = &source.format { + request = request.with_format(format.clone()); + } + if let Some(media_type) = &source.media_type { + request = request.with_media_type(media_type.clone()); + } + IntakeEngine::new() + .intake(&request, store) + .with_context(|| format!("source.intake.failed: '{}'", source.id)) +} + fn resolve_path_relative_to_config(config_path: &Path, value: &str) -> PathBuf { let path = PathBuf::from(value); if path.is_absolute() { @@ -1967,7 +2220,7 @@ fn resolve_target_intent( } apply_exclusions(&mut selected, &profile_exclusions, graph); apply_exclusions(&mut selected, &spec.targets.exclude, graph); - selected.sort_by(|left, right| left.format.to_string().cmp(&right.format.to_string())); + selected.sort_by_key(|item| item.format.to_string()); Ok(selected) } @@ -2566,6 +2819,8 @@ mod tests { members: Vec::new(), media_type: None, format: Some(source_format.to_string()), + sha256: None, + geometry: None, detect: false, immutable: true, }], @@ -2627,8 +2882,7 @@ mod tests { let (available, unavailable) = partition_targets_by_availability(requested, &available_formats, true); - let (planned, used_unavailable_only_fallback) = - planning_targets(&available, &unavailable); + let (planned, used_unavailable_only_fallback) = planning_targets(&available, &unavailable); assert_eq!( available @@ -2663,8 +2917,7 @@ mod tests { let (available, unavailable) = partition_targets_by_availability(requested, &HashSet::new(), true); - let (planned, used_unavailable_only_fallback) = - planning_targets(&available, &unavailable); + let (planned, used_unavailable_only_fallback) = planning_targets(&available, &unavailable); assert!(available.is_empty()); assert_eq!( diff --git a/crates/renderflow-core/src/spec.rs b/crates/renderflow-core/src/spec.rs index 7860db1..7a4be59 100644 --- a/crates/renderflow-core/src/spec.rs +++ b/crates/renderflow-core/src/spec.rs @@ -1,6 +1,7 @@ use std::collections::{BTreeMap, BTreeSet}; use std::fmt; use std::fs; +use std::path::{Component, Path}; use anyhow::{Context, Result}; use serde::{Deserialize, Serialize}; @@ -8,7 +9,7 @@ use serde_json::{json, Value}; use crate::config::Config; use crate::optimization::OptimizationMode; -use crate::publication::PublicationContract; +use crate::publication::{PageGeometry, PublicationContract}; pub const SPEC_V2_ID: &str = "renderflow/v2"; pub const SPEC_V2_SCHEMA_PATH: &str = "schemas/renderflow-v2.schema.json"; @@ -55,6 +56,12 @@ pub struct SourceSpec { pub media_type: Option, #[serde(default)] pub format: Option, + /// Expected payload digest for an explicitly ordered collection member. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub sha256: Option, + /// Declared page geometry; part of collection identity when present. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub geometry: Option, #[serde(default = "default_true")] pub detect: bool, #[serde(default = "default_true")] @@ -715,10 +722,19 @@ impl SpecV2 { for (index, source) in self.sources.iter().enumerate() { if source.kind == SourceKind::Collection { + let mut members = BTreeSet::new(); for (member_index, member) in source.members.iter().enumerate() { + let path = format!("$.sources[{index}].members[{member_index}]"); + if !members.insert(member) { + diagnostics.push(SpecDiagnostic::new( + path.clone(), + "collection.member.duplicate", + format!("collection member '{member}' appears more than once"), + )); + } if !source_ids.contains(member) { diagnostics.push(SpecDiagnostic::new( - format!("$.sources[{index}].members[{member_index}]"), + path.clone(), "collection.member.unknown", format!( "collection member '{member}' does not match a declared source id" @@ -727,11 +743,76 @@ impl SpecV2 { } if member == &source.id { diagnostics.push(SpecDiagnostic::new( - format!("$.sources[{index}].members[{member_index}]"), + path.clone(), "collection.member.self_reference", "a collection cannot contain itself", )); } + if let Some(member_source) = self.sources.iter().find(|item| &item.id == member) + { + if member_source.kind != SourceKind::Artifact { + diagnostics.push(SpecDiagnostic::new( + path.clone(), + "collection.member.nested", + "collection members must be artifact sources", + )); + } + if member_source.path.is_none() || member_source.uri.is_some() { + diagnostics.push(SpecDiagnostic::new( + path.clone(), + "collection.member.locator", + "collection members require a local root-relative path", + )); + } + if let Some(locator) = &member_source.path { + if locator.is_empty() + || !Path::new(locator) + .components() + .all(|component| matches!(component, Component::Normal(_))) + { + diagnostics.push(SpecDiagnostic::new( + format!("$.sources[{index}].members[{member_index}]"), + "collection.member.locator", + "collection member paths must be root-relative without traversal", + )); + } + } + if member_source.format.is_none() || member_source.media_type.is_none() { + diagnostics.push(SpecDiagnostic::new( + path.clone(), + "collection.member.media", + "collection members require explicit format and media_type", + )); + } + if !member_source.sha256.as_ref().is_some_and(|digest| { + digest.len() == 64 + && digest.bytes().all(|byte| { + byte.is_ascii_hexdigit() && !byte.is_ascii_uppercase() + }) + }) { + diagnostics.push(SpecDiagnostic::new( + path, + "collection.member.digest", + "collection members require a lowercase 64-character sha256 digest", + )); + } + if member_source.geometry.as_ref().is_some_and(|geometry| { + !geometry.width.is_finite() + || geometry.width <= 0.0 + || !geometry.height.is_finite() + || geometry.height <= 0.0 + || [geometry.margin, geometry.bleed, geometry.safe_area] + .into_iter() + .flatten() + .any(|value| !value.is_finite() || value < 0.0) + }) { + diagnostics.push(SpecDiagnostic::new( + format!("$.sources[{index}].members[{member_index}]"), + "collection.member.geometry", + "collection member geometry requires finite positive dimensions and nonnegative bounds", + )); + } + } } } } @@ -1296,6 +1377,8 @@ pub(crate) fn migrate_v1_config(config: &Config) -> SpecV2 { members: Vec::new(), media_type: None, format: Some(legacy_source_format(config)), + sha256: None, + geometry: None, detect: config.input_format.is_none(), immutable: true, }], @@ -1466,6 +1549,8 @@ pub fn json_schema() -> Value { "members": {"type": "array", "items": {"$ref": "#/$defs/stableId"}, "default": []}, "media_type": {"type": ["string", "null"]}, "format": {"type": ["string", "null"]}, + "sha256": {"type": ["string", "null"], "pattern": "^[0-9a-f]{64}$"}, + "geometry": {"anyOf": [{"$ref": "#/$defs/pageGeometry"}, {"type": "null"}]}, "detect": {"type": "boolean", "default": true}, "immutable": {"const": true, "default": true} } diff --git a/crates/renderflow-core/src/super_resolution.rs b/crates/renderflow-core/src/super_resolution.rs index 3d4b5f2..ae2300d 100644 --- a/crates/renderflow-core/src/super_resolution.rs +++ b/crates/renderflow-core/src/super_resolution.rs @@ -1117,6 +1117,8 @@ mod tests { members: Vec::new(), media_type: Some("image/png".to_string()), format: Some("png".to_string()), + sha256: None, + geometry: None, detect: false, immutable: true, }], diff --git a/crates/renderflow-core/src/transforms/yaml_loader.rs b/crates/renderflow-core/src/transforms/yaml_loader.rs index a33d3cd..a2a6ce0 100644 --- a/crates/renderflow-core/src/transforms/yaml_loader.rs +++ b/crates/renderflow-core/src/transforms/yaml_loader.rs @@ -642,6 +642,10 @@ pub fn build_graph_executor_and_tools_from_str( let config: YamlTransformConfig = serde_yaml_ng::from_str(yaml).context("Failed to parse YAML transform config")?; + // Bind the complete registry configuration, including argv and provider + // choices, into canonical plan and checkpoint identity without exposing + // potentially sensitive transform settings in public evidence. + let registry_digest = crate::evidence::sha256_text(yaml).value; let tool_registry = parse_tool_registry_from_str(yaml)?; let mut graph = TransformGraph::new(); let mut executor = DagExecutor::new(); @@ -667,7 +671,8 @@ pub fn build_graph_executor_and_tools_from_str( graph.add_transform( TransformEdge::with_input_kind(from, to, def.cost, def.quality, input_kind) .with_provider(provider.to_string(), capability.to_string()) - .with_evidence("transform_id", def.name.clone()), + .with_evidence("transform_id", def.name.clone()) + .with_evidence("registry_sha256", registry_digest.clone()), ); if def.is_collection() { diff --git a/crates/renderflow-core/tests/fixtures/ordered-collection/README.md b/crates/renderflow-core/tests/fixtures/ordered-collection/README.md new file mode 100644 index 0000000..63ac96b --- /dev/null +++ b/crates/renderflow-core/tests/fixtures/ordered-collection/README.md @@ -0,0 +1,18 @@ +# Ordered collection fixture + +Two tiny HTML fragments in `.md` files exercise ordered, immutable local +artifact sources without depending on a publication provider. `aggregate.py` +is a deterministic test-only collection transform that writes a valid HTML +document. The integration tests copy these files to a temporary root, compute +the exact declared SHA-256 digests, and drive the public SDK and CLI. + +`ordered_collection.rs` also generates a 44-member collection from a small +text template. It does not commit 44 binaries. This checks order and plan +identity at realistic page counts before PDF and EPUB exporters exist. + +Run with: + +```bash +cargo test --package renderflow --test ordered_collection --locked +cargo test --package renderflow-cli --test ordered_collection_cli --locked +``` diff --git a/crates/renderflow-core/tests/fixtures/ordered-collection/aggregate.py b/crates/renderflow-core/tests/fixtures/ordered-collection/aggregate.py new file mode 100644 index 0000000..b2e4e4d --- /dev/null +++ b/crates/renderflow-core/tests/fixtures/ordered-collection/aggregate.py @@ -0,0 +1,20 @@ +"""Synthetic deterministic collection transform; no publication exporter.""" + +from pathlib import Path +import sys + + +def main() -> None: + output = Path(sys.argv[1]) + members = [Path(value) for value in sys.argv[2:]] + if not members: + raise SystemExit("at least one member required") + output.write_bytes( + b"\n\n" + + b"\n".join(member.read_bytes() for member in members) + + b"\n\n" + ) + + +if __name__ == "__main__": + main() diff --git a/crates/renderflow-core/tests/fixtures/ordered-collection/page-001.md b/crates/renderflow-core/tests/fixtures/ordered-collection/page-001.md new file mode 100644 index 0000000..ef679bd --- /dev/null +++ b/crates/renderflow-core/tests/fixtures/ordered-collection/page-001.md @@ -0,0 +1 @@ +

First page

diff --git a/crates/renderflow-core/tests/fixtures/ordered-collection/page-002.md b/crates/renderflow-core/tests/fixtures/ordered-collection/page-002.md new file mode 100644 index 0000000..78672e3 --- /dev/null +++ b/crates/renderflow-core/tests/fixtures/ordered-collection/page-002.md @@ -0,0 +1 @@ +

Second page

diff --git a/crates/renderflow-core/tests/ordered_collection.rs b/crates/renderflow-core/tests/ordered_collection.rs new file mode 100644 index 0000000..e44dba5 --- /dev/null +++ b/crates/renderflow-core/tests/ordered_collection.rs @@ -0,0 +1,276 @@ +use std::fs; +use std::path::{Path, PathBuf}; + +use renderflow::evidence::{sha256_serialized, RunState}; +use renderflow::planning::{execute, resolve, PlanningRequest}; +use renderflow::spec::validate_spec_str; +use sha2::{Digest, Sha256}; +use tempfile::TempDir; + +const FIRST: &[u8] = include_bytes!("fixtures/ordered-collection/page-001.md"); +const SECOND: &[u8] = include_bytes!("fixtures/ordered-collection/page-002.md"); +const AGGREGATE: &str = include_str!("fixtures/ordered-collection/aggregate.py"); + +fn setup(count: usize) -> (TempDir, PathBuf, Vec>) { + let dir = tempfile::tempdir().unwrap(); + let mut contents = Vec::new(); + let mut sources = String::new(); + let mut ids = Vec::new(); + for index in 0..count { + let id = format!("source.page{index:03}"); + let path = format!("page-{index:03}.md"); + let bytes = if count == 2 { + if index == 0 { + FIRST.to_vec() + } else { + SECOND.to_vec() + } + } else { + format!("
Page {index}
\n").into_bytes() + }; + fs::write(dir.path().join(&path), &bytes).unwrap(); + let digest = format!("{:x}", Sha256::digest(&bytes)); + sources.push_str(&format!( + " - id: {id}\n path: {path}\n format: markdown\n media_type: text/markdown\n sha256: \"{digest}\"\n geometry: {{ width: 210, height: 297, unit: mm }}\n" + )); + contents.push(bytes); + ids.push(id); + } + sources.push_str(&format!( + " - id: source.pages\n kind: collection\n members: [{}]\n", + ids.join(", ") + )); + let script = dir.path().join("aggregate.py"); + fs::write(&script, AGGREGATE).unwrap(); + fs::write(dir.path().join("transforms.yaml"), format!( + "transforms:\n - name: synthetic.ordered-pages\n program: python3\n args: [\"{}\", \"{{output}}\", \"{{inputs}}\"]\n input_kind: collection\n from: markdown\n to: html\n cost: 1.0\n quality: 1.0\n", + script.display() + )).unwrap(); + let config = dir.path().join("renderflow.yaml"); + fs::write(&config, format!( + "schema: renderflow/v2\nsources:\n{sources}targets:\n exact:\n - id: target.web\n role: web\n format: html\ntransforms: transforms.yaml\noutput:\n bundle_root: \"{}\"\n naming_template: \"{{source.id}}/{{target.role}}.{{ext}}\"\n", + dir.path().join("dist").display() + )).unwrap(); + (dir, config, contents) +} + +fn plan_digest(path: &Path) -> String { + let resolved = resolve(PlanningRequest::from_path(path)).unwrap(); + sha256_serialized(resolved.plan()).unwrap().value +} + +#[test] +fn ordered_collection_executes_with_complete_lineage_and_immutable_sources() { + let (dir, config, original) = setup(2); + let resolved = resolve(PlanningRequest::from_path(&config)).unwrap(); + let collection = resolved.plan().source_collection.as_ref().unwrap(); + assert!(resolved.plan().source_artifact.is_none()); + assert_eq!(collection.source_id, "source.pages"); + assert_eq!( + collection + .members + .iter() + .map(|member| member.source_id.as_str()) + .collect::>(), + ["source.page000", "source.page001"] + ); + assert_eq!(collection.members[0].locator, "page-000.md"); + assert_eq!( + collection.members[0].geometry.as_ref().unwrap().width, + 210.0 + ); + let dry = execute(resolved, true).unwrap(); + assert_eq!(dry.run_manifest.state, RunState::Planned); + assert!(dry.manifest_path.is_none()); + let result = execute(resolve(PlanningRequest::from_path(&config)).unwrap(), false).unwrap(); + assert_eq!( + result.run_manifest.state, + RunState::Complete, + "{:?}", + result.run_manifest.diagnostics + ); + assert_eq!(result.outputs.len(), 1); + let output = fs::read_to_string(&result.outputs[0]).unwrap(); + assert!(output.find("First page").unwrap() < output.find("Second page").unwrap()); + let evidence = &result.run_manifest.artifact_manifest.artifacts; + assert_eq!( + evidence + .iter() + .filter(|item| item.lifecycle == renderflow::evidence::ArtifactRole::Source) + .count(), + 2 + ); + assert_eq!(evidence[0].metadata["renderflow.collection.index"], 0); + assert_eq!(evidence[1].metadata["renderflow.collection.index"], 1); + assert_eq!( + evidence + .iter() + .find(|item| item.role == "web") + .unwrap() + .sources, + vec![ + evidence[0].artifact_id.clone(), + evidence[1].artifact_id.clone() + ] + ); + for (index, bytes) in original.iter().enumerate() { + assert_eq!( + &fs::read(dir.path().join(format!("page-{index:03}.md"))).unwrap(), + bytes + ); + } + assert!(Path::new(result.manifest_path.as_ref().unwrap()).is_file()); + assert!(dir + .path() + .join(format!( + ".renderflow/canonical-cache-{}.json", + result.run_manifest.execution_plan_digest.value + )) + .is_file()); +} + +#[test] +fn collection_plan_binds_order_ids_geometry_configuration_and_larger_recipe() { + let (dir, config, _) = setup(2); + let original = fs::read_to_string(&config).unwrap(); + let first = plan_digest(&config); + assert_eq!(first, plan_digest(&config)); + for changed in [ + original.replace( + "source.page000, source.page001", + "source.page001, source.page000", + ), + original.replace("source.page000", "source.front"), + original.replace("width: 210", "width: 211"), + original.replace("role: web", "role: alternate"), + ] { + fs::write(&config, changed).unwrap(); + assert_ne!(first, plan_digest(&config)); + } + fs::write(&config, original).unwrap(); + let registry_path = dir.path().join("transforms.yaml"); + let registry = fs::read_to_string(®istry_path).unwrap(); + fs::write( + ®istry_path, + registry.replace( + "\"{output}\"", + "\"--different-configuration\", \"{output}\"", + ), + ) + .unwrap(); + assert_ne!( + first, + plan_digest(&config), + "transform argv must enter plan identity" + ); + let (_large_dir, large_config, _) = setup(44); + let resolved = resolve(PlanningRequest::from_path(large_config)).unwrap(); + assert_eq!( + resolved + .plan() + .source_collection + .as_ref() + .unwrap() + .members + .len(), + 44 + ); +} + +#[test] +fn collection_refuses_duplicate_missing_escape_digest_media_and_stale_bytes() { + let (dir, config, _) = setup(2); + let original = fs::read_to_string(&config).unwrap(); + let duplicate = original.replace( + "source.page000, source.page001", + "source.page000, source.page000", + ); + let report = validate_spec_str(&duplicate); + assert!(report + .diagnostics + .iter() + .any(|item| item.code == "collection.member.duplicate")); + let invalid_geometry = original.replace("width: 210", "width: 0"); + assert!(validate_spec_str(&invalid_geometry) + .diagnostics + .iter() + .any(|item| item.code == "collection.member.geometry")); + let ambiguous = original.replace( + " - id: source.pages\n", + " - id: source.unreferenced\n path: other.md\n - id: source.pages\n", + ); + fs::write(&config, ambiguous).unwrap(); + assert!(format!( + "{:#}", + resolve(PlanningRequest::from_path(&config)).err().unwrap() + ) + .contains("collection.source.ambiguous")); + for (spec, code) in [ + ( + original.replace("page-000.md", "missing.md"), + "collection.member.missing", + ), + ( + original.replace("page-000.md", "../page-000.md"), + "collection.member.locator", + ), + ( + original.replace("text/markdown", "application/pdf"), + "collection.member.unsupported_media", + ), + ( + original.replace("sha256: \"", "sha256: \"f"), + "collection.member.digest", + ), + ] { + fs::write(&config, spec).unwrap(); + let error = resolve(PlanningRequest::from_path(&config)).err().unwrap(); + assert!(format!("{error:#}").contains(code), "{code}: {error:#}"); + } + fs::write(&config, original).unwrap(); + let resolved = resolve(PlanningRequest::from_path(&config)).unwrap(); + fs::write(dir.path().join("page-001.md"), b"changed after planning").unwrap(); + let result = execute(resolved, false).unwrap(); + assert_eq!(result.run_manifest.state, RunState::Failed); + assert!(result.outputs.is_empty()); + assert!(result + .run_manifest + .diagnostics + .iter() + .any(|item| item.code == "execution.source_changed_after_intake")); +} + +#[cfg(unix)] +#[test] +fn collection_refuses_symlink_substitution_before_planning_and_execution() { + use std::os::unix::fs::symlink; + let (dir, config, _) = setup(2); + let original = fs::read_to_string(&config).unwrap(); + let alias = dir.path().join("alias.md"); + symlink(dir.path().join("page-000.md"), &alias).unwrap(); + fs::write(&config, original.replace("page-000.md", "alias.md")).unwrap(); + assert!(format!( + "{:#}", + resolve(PlanningRequest::from_path(&config)).err().unwrap() + ) + .contains("collection.member.symlink")); + fs::write(&config, original).unwrap(); + let resolved = resolve(PlanningRequest::from_path(&config)).unwrap(); + fs::rename( + dir.path().join("page-000.md"), + dir.path().join("retained.md"), + ) + .unwrap(); + symlink( + dir.path().join("retained.md"), + dir.path().join("page-000.md"), + ) + .unwrap(); + let result = execute(resolved, false).unwrap(); + assert_eq!(result.run_manifest.state, RunState::Failed); + assert!(result + .run_manifest + .diagnostics + .iter() + .any(|item| item.message.contains("collection.member.symlink"))); +} diff --git a/docs/user-guide/configuration.md b/docs/user-guide/configuration.md index 0c7a1a9..4d1de93 100644 --- a/docs/user-guide/configuration.md +++ b/docs/user-guide/configuration.md @@ -8,7 +8,7 @@ Renderflow supports two explicit configuration contracts: Renderflow never silently reinterprets a declared schema version. Unversioned files are treated as v1 compatibility files; unsupported declared schema identifiers are rejected actionably. !!! important - Issue #353 defines the v2 intent contract, validation, migration, and generated schema. The canonical planner/executor consumes this model in the follow-up unification work tracked by #354. Existing v1 builds remain backward compatible in the meantime. + The canonical planner/executor consumes v1 and v2 through the same lifecycle. Explicit ordered collections are supported for homogeneous local members through a registered collection-input transform. See [Ordered Collections](ordered-collections.md) for the supported boundary. ## Spec v2 @@ -36,6 +36,7 @@ renderflow spec schema --output schemas/renderflow-v2.schema.json ``` See the generated [Spec v2 Reference](spec-v2-reference.md) for the canonical field matrix and complete example. +The broad example expresses mixed-media intent; executable collections currently require members of one declared format and a collection-input transform. ## Versioned derivative profiles and artifact forests diff --git a/docs/user-guide/ordered-collections.md b/docs/user-guide/ordered-collections.md new file mode 100644 index 0000000..68ab0cf --- /dev/null +++ b/docs/user-guide/ordered-collections.md @@ -0,0 +1,89 @@ +# Ordered collections + +A `renderflow/v2` collection names immutable local artifacts in the exact order a +collection-input transform receives them. The public `renderflow spec validate`, +`renderflow build --dry-run`, `renderflow build`, and Rust +`planning::{resolve, execute}` surfaces use the same canonical plan and run +evidence. This is the collection source boundary for later PDF and EPUB +exporters; this checkpoint does not ship those exporters. + +```yaml +schema: renderflow/v2 +sources: + - id: source.page001 + path: pages/001.md + format: markdown + media_type: text/markdown + sha256: "<64 lowercase hex characters for the exact file bytes>" + geometry: { width: 210, height: 297, unit: mm } + - id: source.page002 + path: pages/002.md + format: markdown + media_type: text/markdown + sha256: "<64 lowercase hex characters for the exact file bytes>" + geometry: { width: 210, height: 297, unit: mm } + - id: source.pages + kind: collection + members: [source.page001, source.page002] +targets: + exact: + - format: html + role: web +transforms: transforms.yaml +output: + bundle_root: dist +``` + +Each member ID must be unique. Its path is relative to the spec directory, +with no absolute path, `..`, `.` component, or symlink in the path. A member +must declare a known common format, a media type supported for that format, +and its exact SHA-256 digest. Geometry is optional at this stage, but if +declared it must have positive dimensions and participates in identity. The +selected collection must name every declared artifact exactly once. Nested or +multiple collection sources are unsupported. No glob or directory discovery +chooses membership for you. + +The transform registry must provide a collection-input edge from the common +member format to a distinct output format: + +```yaml +transforms: + - name: synthetic.ordered-pages + program: python3 + args: ["aggregate.py", "{output}", "{inputs}"] + input_kind: collection + from: markdown + to: html + cost: 1.0 + quality: 1.0 +``` + +The example command is a synthetic fixture, not a publication capability. +The runnable fixture and parameterized 44-member recipe are exercised by +`ordered_collection` and `ordered_collection_cli` tests. A real transform +must provide its own output validation and provider contract. + +Planning imports each source into temporary content-addressed staging and +freezes the ordered member IDs, locators, media types, format, geometry, +digests, sizes, and intake profiles in `plan.source_collection`. The plan +digest names the collection cache namespace and, with the full source-spec +digest, binds checkpoint context and run evidence; the ordered input artifact +list binds transform checkpoints. Execution checks +paths again and imports every member into the durable artifact store before +running a transform. A changed or substituted member produces failed run +evidence and no committed target output. + +Run artifact evidence records each member's `renderflow.source_id`, +`renderflow.collection.id`, `renderflow.collection.index`, locator, and +geometry. Aggregated outputs list input artifact IDs in declared order. When +identical member bytes share one content-addressed artifact ID, the distinct +member IDs and indices remain in the frozen plan and source evidence. The +source files are never written by this lifecycle. Resume accepts only a +checkpoint whose plan, source spec, inputs, and provider evidence remain +compatible; otherwise it recomputes or refuses under the existing checkpoint +rules. + +The currently supported canonical execution path requires all root collection +members to share a format and the first edge to consume a collection. Mixed +media, same-format aggregation output, arbitrary root fan-out, PDF/EPUB +generation, and publication approval are outside this checkpoint. diff --git a/docs/user-guide/spec-v2-reference.md b/docs/user-guide/spec-v2-reference.md index 69254fd..93c51e3 100644 --- a/docs/user-guide/spec-v2-reference.md +++ b/docs/user-guide/spec-v2-reference.md @@ -29,6 +29,7 @@ Spec v2 describes source intent, derivative selection, execution policy, and det | --- | --- | --- | --- | | `detect` | `boolean` | no | `true` | | `format` | `string` / `null` | no | — | +| `geometry` | `object` | no | — | | `id` | `stableId` | yes | — | | `immutable` | `true` | no | `true` | | `kind` | `artifact` / `collection` | no | `"artifact"` | @@ -36,6 +37,7 @@ Spec v2 describes source intent, derivative selection, execution policy, and det | `members` | `array` | no | `[]` | | `path` | `string` / `null` | no | — | | `role` | `string` / `null` | no | — | +| `sha256` | `string` / `null` | no | — | | `uri` | `string` / `null` | no | — | ## Publication contract @@ -131,17 +133,23 @@ sources: role: cover path: assets/cover.png media_type: image/png + format: png + sha256: "0000000000000000000000000000000000000000000000000000000000000000" detect: true - id: source.body role: manuscript path: examples/input.md format: markdown + media_type: text/markdown + sha256: "0000000000000000000000000000000000000000000000000000000000000000" detect: true - id: source.publication role: publication kind: collection + # Illustrative mixed-media intent; current canonical execution requires + # a homogeneous collection and a registered collection-input transform. members: - source.cover - source.body diff --git a/examples/renderflow-v2.yaml b/examples/renderflow-v2.yaml index e20ffa9..082c149 100644 --- a/examples/renderflow-v2.yaml +++ b/examples/renderflow-v2.yaml @@ -5,17 +5,23 @@ sources: role: cover path: assets/cover.png media_type: image/png + format: png + sha256: "0000000000000000000000000000000000000000000000000000000000000000" detect: true - id: source.body role: manuscript path: examples/input.md format: markdown + media_type: text/markdown + sha256: "0000000000000000000000000000000000000000000000000000000000000000" detect: true - id: source.publication role: publication kind: collection + # Illustrative mixed-media intent; current canonical execution requires + # a homogeneous collection and a registered collection-input transform. members: - source.cover - source.body diff --git a/mkdocs.yml b/mkdocs.yml index 470ab02..a7c03b4 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -67,6 +67,7 @@ nav: - User Guide: - Configuration: user-guide/configuration.md - Spec v2 Reference: user-guide/spec-v2-reference.md + - Ordered Collections: user-guide/ordered-collections.md - Supported Formats: user-guide/supported-formats.md - Tool Registry: user-guide/tool-registry.md - Adapter Ecosystem: user-guide/adapter-ecosystem.md diff --git a/schemas/renderflow-v2.schema.json b/schemas/renderflow-v2.schema.json index ba7e7c3..fa83591 100644 --- a/schemas/renderflow-v2.schema.json +++ b/schemas/renderflow-v2.schema.json @@ -893,6 +893,16 @@ "null" ] }, + "geometry": { + "anyOf": [ + { + "$ref": "#/$defs/pageGeometry" + }, + { + "type": "null" + } + ] + }, "id": { "$ref": "#/$defs/stableId" }, @@ -932,6 +942,13 @@ "null" ] }, + "sha256": { + "pattern": "^[0-9a-f]{64}$", + "type": [ + "string", + "null" + ] + }, "uri": { "type": [ "string", diff --git a/tests/cli_tests.rs b/tests/cli_tests.rs index 1e092e7..4177881 100644 --- a/tests/cli_tests.rs +++ b/tests/cli_tests.rs @@ -538,9 +538,7 @@ fn test_all_without_transforms_uses_builtin_capability_registry() { "clean-host plan should preserve unavailable branches: {plan}" ); assert!( - branches - .iter() - .all(|branch| branch["state"] != "selected"), + branches.iter().all(|branch| branch["state"] != "selected"), "clean-host plan must not report unavailable providers as selected: {plan}" ); } diff --git a/tests/fixtures/spec-v2/valid-multi-source.yaml b/tests/fixtures/spec-v2/valid-multi-source.yaml index 97e4714..f65d238 100644 --- a/tests/fixtures/spec-v2/valid-multi-source.yaml +++ b/tests/fixtures/spec-v2/valid-multi-source.yaml @@ -3,10 +3,15 @@ sources: - id: source.cover role: cover path: cover.png + format: png + media_type: image/png + sha256: "0000000000000000000000000000000000000000000000000000000000000000" - id: source.body role: manuscript path: body.md format: markdown + media_type: text/markdown + sha256: "0000000000000000000000000000000000000000000000000000000000000000" - id: source.publication role: publication kind: collection diff --git a/tests/ordered_collection_cli.rs b/tests/ordered_collection_cli.rs new file mode 100644 index 0000000..32bff83 --- /dev/null +++ b/tests/ordered_collection_cli.rs @@ -0,0 +1,51 @@ +use std::fs; +use std::process::Command; + +use renderflow::evidence::sha256_text; + +#[test] +fn public_cli_validates_plans_and_executes_ordered_collection() { + let dir = tempfile::tempdir().unwrap(); + let first = + include_str!("../crates/renderflow-core/tests/fixtures/ordered-collection/page-001.md"); + let second = + include_str!("../crates/renderflow-core/tests/fixtures/ordered-collection/page-002.md"); + let script = dir.path().join("aggregate.py"); + fs::write(dir.path().join("page-001.md"), first).unwrap(); + fs::write(dir.path().join("page-002.md"), second).unwrap(); + fs::write( + &script, + include_str!("../crates/renderflow-core/tests/fixtures/ordered-collection/aggregate.py"), + ) + .unwrap(); + fs::write(dir.path().join("transforms.yaml"), format!( + "transforms:\n - name: synthetic.ordered-pages\n program: python3\n args: [\"{}\", \"{{output}}\", \"{{inputs}}\"]\n input_kind: collection\n from: markdown\n to: html\n cost: 1.0\n quality: 1.0\n", + script.display() + )).unwrap(); + let config = dir.path().join("renderflow.yaml"); + fs::write(&config, format!( + "schema: renderflow/v2\nsources:\n - id: source.first\n path: page-001.md\n format: markdown\n media_type: text/markdown\n sha256: \"{}\"\n - id: source.second\n path: page-002.md\n format: markdown\n media_type: text/markdown\n sha256: \"{}\"\n - id: source.pages\n kind: collection\n members: [source.first, source.second]\ntargets:\n exact:\n - format: html\n role: web\ntransforms: transforms.yaml\noutput:\n bundle_root: dist\n", + sha256_text(first).value, + sha256_text(second).value, + )).unwrap(); + let bin = env!("CARGO_BIN_EXE_renderflow"); + for args in [ + vec!["spec", "validate", "--config", "renderflow.yaml"], + vec!["build", "--config", "renderflow.yaml", "--dry-run"], + vec!["build", "--config", "renderflow.yaml"], + ] { + let output = Command::new(bin) + .current_dir(dir.path()) + .args(&args) + .output() + .unwrap(); + assert!( + output.status.success(), + "{args:?}: {}", + String::from_utf8_lossy(&output.stderr) + ); + } + let output = fs::read_to_string(dir.path().join("dist/source.pages/web.html")).unwrap(); + assert!(output.find("First page").unwrap() < output.find("Second page").unwrap()); + assert!(dir.path().join("dist/renderflow-run.json").is_file()); +}