From 7e0dd20dd31c8b0f6616501eb81758783c687f8b Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 04:05:10 -0700 Subject: [PATCH 01/25] fix(scheduler): box runtime execution intent payload --- CHANGELOG.md | 4 ++ .../runtime_dispatch_candidate_provider.rs | 6 +-- crates/pantograph-scheduler/src/queue.rs | 10 +++- .../pantograph-scheduler/tests/queue_state.rs | 46 ++++++++++++++++++- .../src/scheduler/store_task_results.rs | 5 +- .../src/scheduler/store_tests.rs | 2 +- .../src/scheduler/task_orchestrator.rs | 4 +- .../runtime_branch_batch_execution.rs | 4 +- .../src/workflow/session_execution_api.rs | 6 +-- .../src/workflow/task_execution_worker.rs | 6 +-- .../workflow/tests/task_state_read_model.rs | 2 +- .../2026-10-03-runtime-intent-layout.md | 15 ++++++ 12 files changed, 89 insertions(+), 21 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-runtime-intent-layout.md diff --git a/CHANGELOG.md b/CHANGELOG.md index 4a7c1d64a..21f7f2295 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,6 +14,10 @@ The format is based on Keep a Changelog. canonicalization records. ### Changed +- Rust source API: `SchedulerTaskExecutionIntent::Runtime.task_intent` is now boxed. + Construct runtime variants with `SchedulerTaskExecutionIntent::runtime(intent)`; + direct struct-variant callers must pass `Box::new(intent)`. Serialized JSON and + borrowed `runtime_task_intent()` access remain unchanged. - Root project `README.md` reorganized around install, usage, development, and contribution workflows. - Documentation consolidated around current guides, accepted decisions, audits, and active plans; superseded narration remains available in Git history. diff --git a/crates/pantograph-embedded-runtime/src/runtime_dispatch_candidate_provider.rs b/crates/pantograph-embedded-runtime/src/runtime_dispatch_candidate_provider.rs index 53e74907e..a00456cce 100644 --- a/crates/pantograph-embedded-runtime/src/runtime_dispatch_candidate_provider.rs +++ b/crates/pantograph-embedded-runtime/src/runtime_dispatch_candidate_provider.rs @@ -1636,9 +1636,9 @@ mod tests { node_id: intent.node_id, task_id: intent.task_id, state: SchedulerTaskState::Ready { - execution_intent: pantograph_scheduler::SchedulerTaskExecutionIntent::Runtime { - task_intent: schedulable_intent(Some("cuda:0")), - }, + execution_intent: pantograph_scheduler::SchedulerTaskExecutionIntent::runtime( + schedulable_intent(Some("cuda:0")), + ), }, state_version: 1, last_transition_id: SchedulerTaskStateTransitionId::parse("transition.ready") diff --git a/crates/pantograph-scheduler/src/queue.rs b/crates/pantograph-scheduler/src/queue.rs index 53f5c3a7c..2cc890f44 100644 --- a/crates/pantograph-scheduler/src/queue.rs +++ b/crates/pantograph-scheduler/src/queue.rs @@ -274,7 +274,7 @@ impl SchedulerNonRuntimeTaskIntent { #[non_exhaustive] pub enum SchedulerTaskExecutionIntent { Runtime { - task_intent: SchedulableTaskIntent, + task_intent: Box, }, SourceInput { task_intent: SchedulerSourceInputTaskIntent, @@ -285,6 +285,14 @@ pub enum SchedulerTaskExecutionIntent { } impl SchedulerTaskExecutionIntent { + /// Wraps a runtime intent without changing its validation or wire representation. + #[must_use] + pub fn runtime(task_intent: SchedulableTaskIntent) -> Self { + Self::Runtime { + task_intent: Box::new(task_intent), + } + } + #[must_use] pub fn runtime_task_intent(&self) -> Option<&SchedulableTaskIntent> { match self { diff --git a/crates/pantograph-scheduler/tests/queue_state.rs b/crates/pantograph-scheduler/tests/queue_state.rs index dbe2904b3..766d2814e 100644 --- a/crates/pantograph-scheduler/tests/queue_state.rs +++ b/crates/pantograph-scheduler/tests/queue_state.rs @@ -12,6 +12,50 @@ use pantograph_scheduler::{ SCHEDULABLE_TASK_INTENT_CONTRACT_VERSION, SCHEDULER_TASK_STATE_CONTRACT_VERSION, }; +#[test] +fn boxed_runtime_intent_preserves_exact_json_and_borrowed_payload() { + let raw = task_intent("run.001", "task.001"); + let execution = SchedulerTaskExecutionIntent::runtime(raw.clone()); + let expected = serde_json::json!({ "execution_kind": "runtime", "task_intent": raw }); + assert_eq!( + serde_json::to_value(&execution).expect("serialize runtime intent"), + expected + ); + let decoded: SchedulerTaskExecutionIntent = + serde_json::from_value(expected).expect("decode existing wire shape"); + assert_eq!(decoded, execution); + assert_eq!(decoded.runtime_task_intent(), Some(&raw)); + assert!( + std::mem::size_of::() + < std::mem::size_of::() + ); +} + +#[test] +fn boxed_runtime_intent_retains_validation_and_task_correlation() { + let valid = task_intent("run.001", "task.001"); + ValidatedSchedulerTaskStateRecord::try_from(task_record_with_state(ready_state(valid.clone()))) + .expect("valid runtime intent remains accepted"); + let mut invalid = valid; + invalid.contract_version = 0; + let expected = invalid.validate().expect_err("invalid raw version"); + assert_eq!( + task_record_with_state(ready_state(invalid)) + .validate() + .expect_err("boxed invalid version"), + expected + ); + assert_eq!( + task_record_with_state(ready_state(task_intent("run.other", "task.001"))) + .validate() + .expect_err("mismatched run"), + SchedulerContractError::InvalidField { + field: "workflow_run_id", + reason: "task state workflow run id must match task intent", + }, + ); +} + #[test] fn valid_task_state_transition_fixture_decodes_validates_and_applies() { let transition: SchedulerTaskStateTransition = @@ -643,7 +687,7 @@ fn completed_source_input_state(task_kind: &str) -> SchedulerTaskState { } fn runtime_execution_intent(task_intent: SchedulableTaskIntent) -> SchedulerTaskExecutionIntent { - SchedulerTaskExecutionIntent::Runtime { task_intent } + SchedulerTaskExecutionIntent::runtime(task_intent) } fn source_input_execution_intent(task_kind: &str) -> SchedulerTaskExecutionIntent { diff --git a/crates/pantograph-workflow-service/src/scheduler/store_task_results.rs b/crates/pantograph-workflow-service/src/scheduler/store_task_results.rs index 8bfc9a3b6..e48f8a169 100644 --- a/crates/pantograph-workflow-service/src/scheduler/store_task_results.rs +++ b/crates/pantograph-workflow-service/src/scheduler/store_task_results.rs @@ -1176,9 +1176,8 @@ mod tests { workflow_run_id: &str, task_id: &str, ) -> SchedulerTaskState { - let execution_intent = SchedulerTaskExecutionIntent::Runtime { - task_intent: task_intent(workflow_run_id, task_id), - }; + let execution_intent = + SchedulerTaskExecutionIntent::runtime(task_intent(workflow_run_id, task_id)); match state { SchedulerTaskStateKind::AwaitingInputs => SchedulerTaskState::AwaitingInputs { diagnostics: Vec::new(), diff --git a/crates/pantograph-workflow-service/src/scheduler/store_tests.rs b/crates/pantograph-workflow-service/src/scheduler/store_tests.rs index e9d8ef635..3bcdccfcb 100644 --- a/crates/pantograph-workflow-service/src/scheduler/store_tests.rs +++ b/crates/pantograph-workflow-service/src/scheduler/store_tests.rs @@ -176,7 +176,7 @@ fn scheduler_state( } fn runtime_execution_intent(task_intent: SchedulableTaskIntent) -> SchedulerTaskExecutionIntent { - SchedulerTaskExecutionIntent::Runtime { task_intent } + SchedulerTaskExecutionIntent::runtime(task_intent) } fn scheduler_state_diagnostics() -> Vec { diff --git a/crates/pantograph-workflow-service/src/scheduler/task_orchestrator.rs b/crates/pantograph-workflow-service/src/scheduler/task_orchestrator.rs index 3a4aa2386..2f6217dc2 100644 --- a/crates/pantograph-workflow-service/src/scheduler/task_orchestrator.rs +++ b/crates/pantograph-workflow-service/src/scheduler/task_orchestrator.rs @@ -2718,7 +2718,7 @@ fn initial_task_state( } if let Some(task_intent) = task.schedulable_intent.clone() { Ok(SchedulerTaskState::WaitingDependencyReadiness { - execution_intent: SchedulerTaskExecutionIntent::Runtime { task_intent }, + execution_intent: SchedulerTaskExecutionIntent::runtime(task_intent), }) } else { Ok(awaiting_inputs_state()) @@ -3601,7 +3601,7 @@ fn runtime_execution_intent( )), )); }; - Ok(SchedulerTaskExecutionIntent::Runtime { task_intent }) + Ok(SchedulerTaskExecutionIntent::runtime(task_intent)) } enum MaterializedBindingValue<'a> { diff --git a/crates/pantograph-workflow-service/src/workflow/runtime_branch_batch_execution.rs b/crates/pantograph-workflow-service/src/workflow/runtime_branch_batch_execution.rs index d31c6580a..bb2cf341f 100644 --- a/crates/pantograph-workflow-service/src/workflow/runtime_branch_batch_execution.rs +++ b/crates/pantograph-workflow-service/src/workflow/runtime_branch_batch_execution.rs @@ -3224,9 +3224,7 @@ mod tests { } fn runtime_execution_intent(workflow_run_id: &str) -> SchedulerTaskExecutionIntent { - SchedulerTaskExecutionIntent::Runtime { - task_intent: task_intent_for_run(workflow_run_id), - } + SchedulerTaskExecutionIntent::runtime(task_intent_for_run(workflow_run_id)) } fn prompt_task_result(workflow_run_id: &str, prompt_text: &str) -> WorkflowSchedulerTaskResult { diff --git a/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs b/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs index b282e59c3..e8bd263f2 100644 --- a/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs +++ b/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs @@ -2643,8 +2643,8 @@ mod tests { node_id: task.node_id.clone(), task_id: task.task_id.clone(), state: SchedulerTaskState::Completed { - execution_intent: SchedulerTaskExecutionIntent::Runtime { - task_intent: pantograph_scheduler::SchedulableTaskIntent { + execution_intent: SchedulerTaskExecutionIntent::runtime( + pantograph_scheduler::SchedulableTaskIntent { contract_version: pantograph_scheduler::SCHEDULABLE_TASK_INTENT_CONTRACT_VERSION, workflow_id: task.workflow_id.clone(), @@ -2668,7 +2668,7 @@ mod tests { dependency_override_patches: Vec::new(), estimate_hints: Vec::new(), }, - }, + ), }, state_version: 1, last_transition_id: "transition.completed".parse().expect("transition id"), diff --git a/crates/pantograph-workflow-service/src/workflow/task_execution_worker.rs b/crates/pantograph-workflow-service/src/workflow/task_execution_worker.rs index 6c71c09e9..5a15d2c97 100644 --- a/crates/pantograph-workflow-service/src/workflow/task_execution_worker.rs +++ b/crates/pantograph-workflow-service/src/workflow/task_execution_worker.rs @@ -3992,9 +3992,9 @@ mod tests { node_id: SchedulerNodeId::parse(WORKER_BATCH_NODE_ID).expect("node id"), task_id: SchedulerTaskId::parse(WORKER_BATCH_TASK_ID).expect("task id"), state: SchedulerTaskState::Ready { - execution_intent: SchedulerTaskExecutionIntent::Runtime { - task_intent: task_intent_for_run(workflow_run_id), - }, + execution_intent: SchedulerTaskExecutionIntent::runtime(task_intent_for_run( + workflow_run_id, + )), }, state_version: 1, last_transition_id: SchedulerTaskStateTransitionId::parse(format!( diff --git a/crates/pantograph-workflow-service/src/workflow/tests/task_state_read_model.rs b/crates/pantograph-workflow-service/src/workflow/tests/task_state_read_model.rs index e95be212d..f82af0582 100644 --- a/crates/pantograph-workflow-service/src/workflow/tests/task_state_read_model.rs +++ b/crates/pantograph-workflow-service/src/workflow/tests/task_state_read_model.rs @@ -85,7 +85,7 @@ fn state_with_intent( } fn runtime_execution_intent(task_intent: SchedulableTaskIntent) -> SchedulerTaskExecutionIntent { - SchedulerTaskExecutionIntent::Runtime { task_intent } + SchedulerTaskExecutionIntent::runtime(task_intent) } fn scheduler_task_graph(workflow_run_id: &str) -> WorkflowSchedulerTaskGraph { diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-runtime-intent-layout.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-runtime-intent-layout.md new file mode 100644 index 000000000..4f148cb66 --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-runtime-intent-layout.md @@ -0,0 +1,15 @@ +# Boxed runtime intent payload + +## Accepted source-API decision + +Box only the large Runtime.task_intent field in SchedulerTaskExecutionIntent and add runtime(intent) as the preferred constructor. Migrate all ten repository construction sites; the remaining explicit Runtime match only inspects the variant. Borrowed runtime_task_intent access, validation/correlation, and tagged Serde representation remain unchanged. Source consumers directly constructing the public struct variant must now supply Box::new(intent); the changelog records this source-breaking change. Unseen Git/path consumers are not claimed compatible. + +The scheduler crate is publish=false and inherits workspace development version 0.1.0. docs/release.md says no accepted release workflow currently exists; it distinguishes development packaging from release acceptance. This milestone therefore does not invent a crate release/version bump or publish a release artifact. + +## Verification plan and evidence + +New regressions compare exact tagged JSON, deserialize the old wire shape, verify borrowed payload equality, check reduced inline layout, and retain raw validation errors plus exact task/run correlation failure. Existing scheduler suites and workflow-service/embedded-runtime callers remain in hosted qualification. + +Rust formatting and whitespace checks pass locally. No local workspace compilation or executed new Rust test is claimed; source review and fresh hosted scheduler/Clippy/workflow tests are required. Existing aggregate gates are unchanged, and this is an internal layout/source-API repair rather than the future research scheduler redesign. + +Independent source review accepted tree e3f247a05873d68632044d1078b7490b76529fbf. Fresh hosted compilation/runtime qualification remains required. The layout assertion checks inline size only and is not evidence of measured execution-performance improvement. From 5218c1ace1cde44efbfb9f6202b910ea8462c08c Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 04:16:46 -0700 Subject: [PATCH 02/25] fix(media): accept named conversion result input --- .github/workflows/quality-gates.yml | 3 + CHANGELOG.md | 3 + crates/pantograph-media-conversion/src/lib.rs | 207 ++++++++++++++---- .../reports/2026-10-03-media-result-input.md | 11 + 4 files changed, 185 insertions(+), 39 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-media-result-input.md diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml index 6481af342..9c4851e4c 100644 --- a/.github/workflows/quality-gates.yml +++ b/.github/workflows/quality-gates.yml @@ -190,6 +190,9 @@ jobs: - name: Check scheduler standard-trait accessor lint run: cargo clippy -p pantograph-scheduler --all-targets -- -D clippy::should_implement_trait + - name: Run media conversion contract tests + run: cargo test -p pantograph-media-conversion + - name: Run node-engine tests run: cargo test -p node-engine --lib diff --git a/CHANGELOG.md b/CHANGELOG.md index 21f7f2295..e293441d3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,6 +14,9 @@ The format is based on Keep a Changelog. canonicalization records. ### Changed +- Rust source API: `MediaConversionResult::try_new` now accepts one + `MediaConversionResultInput` with the same eight named fields instead of eight + positional arguments. Result fields and validation behavior are unchanged. - Rust source API: `SchedulerTaskExecutionIntent::Runtime.task_intent` is now boxed. Construct runtime variants with `SchedulerTaskExecutionIntent::runtime(intent)`; direct struct-variant callers must pass `Box::new(intent)`. Serialized JSON and diff --git a/crates/pantograph-media-conversion/src/lib.rs b/crates/pantograph-media-conversion/src/lib.rs index 70449e236..8ca4089d2 100644 --- a/crates/pantograph-media-conversion/src/lib.rs +++ b/crates/pantograph-media-conversion/src/lib.rs @@ -1299,16 +1299,16 @@ where } let command_id = request.target.format_id.as_str().to_string(); - MediaConversionResult::try_new( - request.conversion_id, - MediaConversionStatus::Converted, - self.converter.target_media_type.clone(), - request.target, + MediaConversionResult::try_new(MediaConversionResultInput { + conversion_id: request.conversion_id, + status: MediaConversionStatus::Converted, + media_type: self.converter.target_media_type.clone(), + target: request.target, command_id, - output.stdout, - vec![self.converter.dependency.clone()], - output.stderr_summary, - ) + body: output.stdout, + dependencies: vec![self.converter.dependency.clone()], + stderr_summary: output.stderr_summary, + }) } } @@ -1324,17 +1324,33 @@ pub struct MediaConversionResult { pub stderr_summary: Option, } +/// Named, unvalidated input to the result constructor. +/// +/// This preserves the result field representation while making construction explicit. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct MediaConversionResultInput { + pub conversion_id: MediaConversionId, + pub status: MediaConversionStatus, + pub media_type: MediaType, + pub target: MediaConversionTarget, + pub command_id: String, + pub body: Vec, + pub dependencies: Vec, + pub stderr_summary: Option, +} + impl MediaConversionResult { - pub fn try_new( - conversion_id: MediaConversionId, - status: MediaConversionStatus, - media_type: MediaType, - target: MediaConversionTarget, - command_id: String, - body: Vec, - dependencies: Vec, - stderr_summary: Option, - ) -> Result { + pub fn try_new(input: MediaConversionResultInput) -> Result { + let MediaConversionResultInput { + conversion_id, + status, + media_type, + target, + command_id, + body, + dependencies, + stderr_summary, + } = input; if body.is_empty() { return Err(MediaConversionError::MissingField { field: "body" }); } @@ -2027,22 +2043,135 @@ mod tests { .to_string(), }; - let result = MediaConversionResult::try_new( - id("conversion-a"), - MediaConversionStatus::Converted, - "image/jpeg".parse().expect("media type"), - target(), - "oiiotool_jpg".to_string(), - vec![1, 2, 3], - vec![dependency.clone()], - Some("bounded stderr".to_string()), - ) + let result = MediaConversionResult::try_new(MediaConversionResultInput { + conversion_id: id("conversion-a"), + status: MediaConversionStatus::Converted, + media_type: "image/jpeg".parse().expect("media type"), + target: target(), + command_id: "oiiotool_jpg".to_string(), + body: vec![1, 2, 3], + dependencies: vec![dependency.clone()], + stderr_summary: Some("bounded stderr".to_string()), + }) .expect("result"); assert_eq!(result.dependencies, vec![dependency]); assert_eq!(result.status, MediaConversionStatus::Converted); } + fn valid_result_input() -> MediaConversionResultInput { + MediaConversionResultInput { + conversion_id: id("conversion-a"), + status: MediaConversionStatus::Converted, + media_type: "image/jpeg".parse().expect("media type"), + target: target(), + command_id: "oiiotool_jpg".to_string(), + body: vec![1, 2, 3], + dependencies: vec![MediaConversionDependencyAttribution { + dependency_id: ManagedMediaDependencyId::Oiiotool, + version: id("2.5.18"), + lease_id: id("lease-1"), + lease_holder: "lease holder".to_string(), + }], + stderr_summary: Some("bounded stderr".to_string()), + } + } + + #[test] + fn named_result_input_preserves_all_fields_and_existing_normalization() { + let mut input = valid_result_input(); + input.command_id = " oiiotool_jpg ".to_string(); + input.stderr_summary = Some(" bounded stderr ".to_string()); + input.dependencies[0].lease_holder = " lease holder ".to_string(); + let result = MediaConversionResult::try_new(input.clone()).expect("valid named input"); + assert_eq!( + result, + MediaConversionResult { + conversion_id: input.conversion_id, + status: input.status, + media_type: input.media_type, + target: input.target, + command_id: "oiiotool_jpg".to_string(), + body: input.body, + dependencies: input.dependencies, + stderr_summary: input.stderr_summary, + } + ); + } + + #[test] + fn result_fields_round_trip_through_named_input() { + let result = MediaConversionResult::try_new(valid_result_input()).expect("result"); + let round_trip = MediaConversionResult::try_new(MediaConversionResultInput { + conversion_id: result.conversion_id.clone(), + status: result.status, + media_type: result.media_type.clone(), + target: result.target.clone(), + command_id: result.command_id.clone(), + body: result.body.clone(), + dependencies: result.dependencies.clone(), + stderr_summary: result.stderr_summary.clone(), + }) + .expect("result round trip"); + assert_eq!(round_trip, result); + } + + #[test] + fn named_result_input_preserves_validation_order_and_errors() { + let mut input = valid_result_input(); + input.body.clear(); + input.command_id = "bad command".to_string(); + input.stderr_summary = Some("bad\nsummary".to_string()); + input.dependencies[0].lease_holder = "bad\nlease".to_string(); + assert_eq!( + MediaConversionResult::try_new(input.clone()).expect_err("body first"), + MediaConversionError::MissingField { field: "body" } + ); + input.body.push(1); + assert_eq!( + MediaConversionResult::try_new(input.clone()).expect_err("command second"), + MediaConversionError::InvalidIdentifier { + field: "command_id" + } + ); + input.command_id = "valid".to_string(); + assert_eq!( + MediaConversionResult::try_new(input.clone()).expect_err("stderr third"), + MediaConversionError::InvalidText { + field: "stderr_summary" + } + ); + input.stderr_summary = None; + assert_eq!( + MediaConversionResult::try_new(input).expect_err("lease last"), + MediaConversionError::InvalidText { + field: "dependency_lease_holder" + } + ); + } + + #[test] + fn named_result_input_preserves_text_bounds() { + let mut input = valid_result_input(); + input.stderr_summary = Some("x".repeat(MAX_ERROR_SUMMARY_LEN + 1)); + assert_eq!( + MediaConversionResult::try_new(input.clone()).expect_err("stderr bound"), + MediaConversionError::FieldTooLong { + field: "stderr_summary", + max_len: MAX_ERROR_SUMMARY_LEN + } + ); + input.stderr_summary = None; + input.dependencies[0].lease_holder = "x".repeat(MAX_LEASE_HOLDER_LEN + 1); + assert_eq!( + MediaConversionResult::try_new(input).expect_err("lease bound"), + MediaConversionError::FieldTooLong { + field: "dependency_lease_holder", + max_len: MAX_LEASE_HOLDER_LEN + } + ); + } + #[test] fn source_and_result_reject_empty_bodies() { let source_error = MediaConversionSource::try_new( @@ -2056,16 +2185,16 @@ mod tests { MediaConversionError::MissingField { field: "body" } )); - let result_error = MediaConversionResult::try_new( - id("conversion-a"), - MediaConversionStatus::Converted, - "image/jpeg".parse().expect("media type"), - target(), - "oiiotool_jpg".to_string(), - Vec::new(), - Vec::new(), - None, - ) + let result_error = MediaConversionResult::try_new(MediaConversionResultInput { + conversion_id: id("conversion-a"), + status: MediaConversionStatus::Converted, + media_type: "image/jpeg".parse().expect("media type"), + target: target(), + command_id: "oiiotool_jpg".to_string(), + body: Vec::new(), + dependencies: Vec::new(), + stderr_summary: None, + }) .expect_err("empty result body"); assert!(matches!( result_error, diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-media-result-input.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-media-result-input.md new file mode 100644 index 000000000..858775f23 --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-media-result-input.md @@ -0,0 +1,11 @@ +# Named media conversion result input + +The next aggregate warning-deny Clippy failure after the scheduler layout repair was the eight-argument MediaConversionResult::try_new constructor. Replace its positional arguments with MediaConversionResultInput carrying exactly the existing eight fields. The constructor destructures that input and retains the existing validation statements, ordering, normalization, errors, and result representation. Migrate all three repository callers: the managed executor and two existing tests. + +This is an explicit public Rust construction change despite the crate being publish=false at workspace development version 0.1.0. The changelog records the new call shape; unseen external Git/path consumers are not assumed compatible. No serialized DTO, conversion routing, process execution, or dependency policy changes are included. + +Four regressions cover all successful fields and existing command-only normalization, result-field round-trip, ordered failures for body/command/stderr/dependency holder, and text bounds. Existing executor success, failure, timeout and cancellation tests remain. The focused hosted Rust job now runs the media conversion crate tests explicitly. + +Formatting and whitespace checks pass locally. No local workspace compilation or executed new Rust tests are claimed. Independent source review and fresh hosted media/scheduler/aggregate Clippy qualification are pending. Existing gates remain intact, with no lint suppression. + +Independent source review accepted final tree 443829480fb96b4164eb068adfa51ee9e8d14d92. This confirms the bounded source/API migration and regression coverage; actual Rust execution remains hosted qualification. From cac290b6b5087c494a41a1945a591aacf55c7f65 Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 04:29:59 -0700 Subject: [PATCH 03/25] fix(inference): preserve behavior while clearing annotation and style lints --- .github/workflows/quality-gates.yml | 6 ++ crates/inference/src/backend/compatibility.rs | 49 +++++++++------ crates/inference/src/execution_telemetry.rs | 1 - crates/inference/src/gateway.rs | 4 +- .../inference/src/image_generation_planner.rs | 1 - crates/inference/src/model_contracts.rs | 19 +++--- crates/inference/src/server.rs | 6 +- crates/inference/src/server_tests.rs | 62 +++++++++++++++++++ crates/inference/src/types.rs | 26 -------- .../2026-10-03-inference-lint-basics.md | 9 +++ 10 files changed, 123 insertions(+), 60 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-inference-lint-basics.md diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml index 9c4851e4c..4aef59bdb 100644 --- a/.github/workflows/quality-gates.yml +++ b/.github/workflows/quality-gates.yml @@ -193,6 +193,12 @@ jobs: - name: Run media conversion contract tests run: cargo test -p pantograph-media-conversion + - name: Run inference compatibility and contract tests + run: | + cargo test -p inference --lib backend::compatibility::tests + cargo test -p inference --lib model_contracts::tests + cargo test -p inference --lib server::tests::runtime_matchers_preserve_path_component_comparison + - name: Run node-engine tests run: cargo test -p node-engine --lib diff --git a/crates/inference/src/backend/compatibility.rs b/crates/inference/src/backend/compatibility.rs index da7be876d..f879a94d8 100644 --- a/crates/inference/src/backend/compatibility.rs +++ b/crates/inference/src/backend/compatibility.rs @@ -181,21 +181,25 @@ impl BackendCapabilities { &mut issues, ); let option_diagnostics = self.option_compatibility_diagnostics(backend_key, &request); - issues.extend(option_diagnostics.iter().filter_map(|diagnostic| { - matches!( - &diagnostic.state, - OptionSupportState::Unsupported | OptionSupportState::Rejected - ) - .then(|| BackendCompatibilityIssue { - kind: BackendCompatibilityIssueKind::UnsupportedOption, - phase: InferenceLifecyclePhase::TaskValidation, - message: diagnostic.message.clone().unwrap_or_else(|| { - format!("option {} is not supported", diagnostic.option_path) + issues.extend( + option_diagnostics + .iter() + .filter(|diagnostic| { + matches!( + &diagnostic.state, + OptionSupportState::Unsupported | OptionSupportState::Rejected + ) + }) + .map(|diagnostic| BackendCompatibilityIssue { + kind: BackendCompatibilityIssueKind::UnsupportedOption, + phase: InferenceLifecyclePhase::TaskValidation, + message: diagnostic.message.clone().unwrap_or_else(|| { + format!("option {} is not supported", diagnostic.option_path) + }), + model_id: Some(request.package_facts.model_ref.model_id.clone()), + path: None, }), - model_id: Some(request.package_facts.model_ref.model_id.clone()), - path: None, - }) - })); + ); let compatible = [task, model_source, preprocessing, postprocessing] .into_iter() .all(|status| status == BackendCompatibilityStatus::Supported) @@ -452,10 +456,7 @@ fn compatibility_report_status_label(report: &BackendCompatibilityReport) -> &'s report.preprocessing, report.postprocessing, ]; - if statuses - .iter() - .any(|status| *status == BackendCompatibilityStatus::Unsupported) - { + if statuses.contains(&BackendCompatibilityStatus::Unsupported) { "rejected" } else { "degraded" @@ -909,6 +910,7 @@ mod tests { Some("llama_cpp"), BackendCompatibilityRequest::new(&task, &package).with_options( BackendCompatibilityOptions { + streaming: true, cache: CacheGenerationOptions { use_cache: Some(true), kv_cache_checkpoint_requested: Some(true), @@ -919,6 +921,17 @@ mod tests { ); assert!(!report.compatible); + assert_eq!( + report + .issues + .iter() + .map(|issue| issue.message.as_str()) + .collect::>(), + vec![ + "option cache.use_cache is not supported by this backend", + "option cache.kv_cache_checkpoint_requested is not supported by this backend", + ], + ); assert!(report.option_diagnostics.iter().any(|diagnostic| { diagnostic.option_path == "cache.use_cache" && matches!(&diagnostic.state, OptionSupportState::Unsupported) diff --git a/crates/inference/src/execution_telemetry.rs b/crates/inference/src/execution_telemetry.rs index d0691c038..0523b0741 100644 --- a/crates/inference/src/execution_telemetry.rs +++ b/crates/inference/src/execution_telemetry.rs @@ -24,7 +24,6 @@ pub struct InferenceExecutionTelemetryScope { } impl InferenceExecutionTelemetryScope { - #[must_use] pub fn new() -> Self { Self::default() } diff --git a/crates/inference/src/gateway.rs b/crates/inference/src/gateway.rs index c8bc03509..fea9f01a4 100644 --- a/crates/inference/src/gateway.rs +++ b/crates/inference/src/gateway.rs @@ -3497,9 +3497,7 @@ fn start_runtime_resource_monitor_for_process( fn finish_runtime_resource_monitor( guard: Option, ) -> Option { - let Some(guard) = guard else { - return None; - }; + let guard = guard?; match guard.finish() { Ok(observation) => Some(observation), Err(error) => { diff --git a/crates/inference/src/image_generation_planner.rs b/crates/inference/src/image_generation_planner.rs index 773afb52f..7008fe941 100644 --- a/crates/inference/src/image_generation_planner.rs +++ b/crates/inference/src/image_generation_planner.rs @@ -101,7 +101,6 @@ impl PlannedImageGenerationLaunchHandoff { &self.artifact_load_target } - #[must_use] pub fn backend_decision(&self) -> &BackendExecutionDecision { &self.backend_decision } diff --git a/crates/inference/src/model_contracts.rs b/crates/inference/src/model_contracts.rs index 795056401..c88a1e4cb 100644 --- a/crates/inference/src/model_contracts.rs +++ b/crates/inference/src/model_contracts.rs @@ -343,22 +343,17 @@ impl TaskModalitySignature { } /// Support tier for a task or backend mapping. -#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)] +#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq, Eq)] #[serde(rename_all = "snake_case")] pub enum SupportTier { Stable, Experimental, Roadmap, Unsupported, + #[default] Unknown, } -impl Default for SupportTier { - fn default() -> Self { - Self::Unknown - } -} - /// Broad task family used for compatibility diagnostics without choosing a /// runtime or scheduler policy. #[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq, Eq)] @@ -2136,7 +2131,6 @@ impl ResolvedModelSource { /// /// This enforces model-source shape invariants without selecting a backend /// or deciding runtime placement. - #[must_use] pub fn validate_for_backend_load(&self) -> Result<(), Vec> { let mut diagnostics = Vec::new(); @@ -2423,6 +2417,15 @@ mod tests { .unwrap_or_else(|| panic!("missing request contract for {task_id:?}")) } + #[test] + fn support_tier_default_retains_unknown_wire_value() { + assert_eq!(SupportTier::default(), SupportTier::Unknown); + assert_eq!( + serde_json::to_value(SupportTier::default()).expect("default serializes"), + serde_json::json!("unknown") + ); + } + #[test] fn task_registry_entries_publish_typed_request_contracts() { let text = registry_contract(InferenceTaskId::TextGeneration); diff --git a/crates/inference/src/server.rs b/crates/inference/src/server.rs index 8925cf983..2de053845 100644 --- a/crates/inference/src/server.rs +++ b/crates/inference/src/server.rs @@ -648,7 +648,7 @@ impl LlamaServer { return false; }; active.mode == LlamaCppRuntimeMode::Inference - && active.model_path == PathBuf::from(model_path) + && active.model_path.as_path() == Path::new(model_path) && active.mmproj_path == mmproj_path.map(PathBuf::from) && active.device == *device && active.context_size == Some(context_size) @@ -669,7 +669,7 @@ impl LlamaServer { return false; }; active.mode == LlamaCppRuntimeMode::Embedding - && active.model_path == PathBuf::from(model_path) + && active.model_path.as_path() == Path::new(model_path) && active.device == *device && active.port == expected_port } @@ -685,7 +685,7 @@ impl LlamaServer { return false; }; active.mode == LlamaCppRuntimeMode::Reranking - && active.model_path == PathBuf::from(model_path) + && active.model_path.as_path() == Path::new(model_path) && active.device == *device && active.port == expected_port } diff --git a/crates/inference/src/server_tests.rs b/crates/inference/src/server_tests.rs index eed60e3ce..08e13b685 100644 --- a/crates/inference/src/server_tests.rs +++ b/crates/inference/src/server_tests.rs @@ -424,3 +424,65 @@ fn assert_arg_pair(args: &[String], name: &str, value: &str) { "expected arg pair {name} {value} in {args:?}" ); } + +#[test] +fn runtime_matchers_preserve_path_component_comparison() { + let device = DeviceConfig { + device: DeviceBackend::Auto, + gpu_layers: -1, + }; + let mut server = LlamaServer::new(); + server.set_test_runtime_state( + ServerMode::SidecarInference { + port: 11434, + model_path: "models/main.gguf".to_string(), + mmproj_path: None, + device: device.clone(), + context_size: 4096, + cpu_threads: None, + batch_size: None, + ubatch_size: None, + }, + true, + ); + assert!(server.matches_inference_runtime( + "models/./main.gguf", + None, + &device, + 4096, + None, + None, + None, + Some(11434) + )); + assert!(!server.matches_inference_runtime( + "models/other.gguf", + None, + &device, + 4096, + None, + None, + None, + Some(11434) + )); + server.set_test_runtime_state( + ServerMode::SidecarEmbedding { + port: 11434, + model_path: "models/main.gguf".to_string(), + device: device.clone(), + }, + true, + ); + assert!(server.matches_embedding_runtime("models/./main.gguf", &device, Some(11434))); + assert!(!server.matches_embedding_runtime("models/other.gguf", &device, Some(11434))); + server.set_test_runtime_state( + ServerMode::SidecarReranking { + port: 11434, + model_path: "models/main.gguf".to_string(), + device: device.clone(), + }, + true, + ); + assert!(server.matches_reranking_runtime("models/./main.gguf", &device, Some(11434))); + assert!(!server.matches_reranking_runtime("models/other.gguf", &device, Some(11434))); +} diff --git a/crates/inference/src/types.rs b/crates/inference/src/types.rs index c2451f46b..e91c5dd96 100644 --- a/crates/inference/src/types.rs +++ b/crates/inference/src/types.rs @@ -922,7 +922,6 @@ pub struct InferenceRequestLifecycleEvent { } impl InferenceRequestLifecycleEvent { - #[must_use] pub fn builder( phase: InferenceLifecyclePhase, kind: InferenceRequestLifecycleEventKind, @@ -950,7 +949,6 @@ pub struct InferenceRequestLifecycleEventContext { } impl InferenceRequestLifecycleEventContext { - #[must_use] pub fn builder( &self, phase: InferenceLifecyclePhase, @@ -980,7 +978,6 @@ pub struct InferenceRequestLifecycleEventBuilder { } impl InferenceRequestLifecycleEventBuilder { - #[must_use] pub fn new( phase: InferenceLifecyclePhase, kind: InferenceRequestLifecycleEventKind, @@ -1015,49 +1012,41 @@ impl InferenceRequestLifecycleEventBuilder { } } - #[must_use] pub fn with_request_id(mut self, request_id: Option) -> Self { self.event.request_id = request_id; self } - #[must_use] pub fn with_phase(mut self, phase: InferenceLifecyclePhase) -> Self { self.event.phase = phase; self } - #[must_use] pub fn with_kind(mut self, kind: InferenceRequestLifecycleEventKind) -> Self { self.event.kind = kind; self } - #[must_use] pub fn with_occurred_at_ms(mut self, occurred_at_ms: u64) -> Self { self.event.occurred_at_ms = occurred_at_ms; self } - #[must_use] pub fn with_task_id(mut self, task_id: Option) -> Self { self.event.task_id = task_id; self } - #[must_use] pub fn with_backend_key(mut self, backend_key: Option) -> Self { self.event.backend_key = backend_key; self } - #[must_use] pub fn with_runtime_id(mut self, runtime_id: Option) -> Self { self.event.runtime_id = runtime_id; self } - #[must_use] pub fn with_selected_runtime_variant_id( mut self, selected_runtime_variant_id: Option, @@ -1066,13 +1055,11 @@ impl InferenceRequestLifecycleEventBuilder { self } - #[must_use] pub fn with_runtime_instance_id(mut self, runtime_instance_id: Option) -> Self { self.event.runtime_instance_id = runtime_instance_id; self } - #[must_use] pub fn with_selected_device_class( mut self, selected_device_class: Option, @@ -1081,7 +1068,6 @@ impl InferenceRequestLifecycleEventBuilder { self } - #[must_use] pub fn with_selected_device_id( mut self, selected_device_id: Option, @@ -1090,7 +1076,6 @@ impl InferenceRequestLifecycleEventBuilder { self } - #[must_use] pub fn with_selected_network_node_id( mut self, selected_network_node_id: Option, @@ -1099,37 +1084,31 @@ impl InferenceRequestLifecycleEventBuilder { self } - #[must_use] pub fn with_model_id(mut self, model_id: Option) -> Self { self.event.model_id = model_id; self } - #[must_use] pub fn with_resolved_artifact_kind(mut self, resolved_artifact_kind: Option) -> Self { self.event.resolved_artifact_kind = resolved_artifact_kind; self } - #[must_use] pub fn with_usage(mut self, usage: Option) -> Self { self.event.usage = usage; self } - #[must_use] pub fn with_cache_handle_id(mut self, cache_handle_id: Option) -> Self { self.event.cache_handle_id = cache_handle_id; self } - #[must_use] pub fn with_artifact_refs(mut self, artifact_refs: Vec) -> Self { self.event.artifact_refs = artifact_refs; self } - #[must_use] pub fn with_resource_observation( mut self, resource_observation: Option, @@ -1138,13 +1117,11 @@ impl InferenceRequestLifecycleEventBuilder { self } - #[must_use] pub fn with_detail(mut self, detail: Option) -> Self { self.event.detail = detail; self } - #[must_use] pub fn with_canonical_error_event_id( mut self, canonical_error_event_id: Option, @@ -1153,7 +1130,6 @@ impl InferenceRequestLifecycleEventBuilder { self } - #[must_use] pub fn with_compatibility_report( mut self, compatibility_report: Option, @@ -1162,7 +1138,6 @@ impl InferenceRequestLifecycleEventBuilder { self } - #[must_use] pub fn with_compatibility_issues( mut self, compatibility_issues: Vec, @@ -1171,7 +1146,6 @@ impl InferenceRequestLifecycleEventBuilder { self } - #[must_use] pub fn with_option_diagnostics( mut self, option_diagnostics: Vec, diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-inference-lint-basics.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-inference-lint-basics.md new file mode 100644 index 000000000..bfb4c5b7f --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-inference-lint-basics.md @@ -0,0 +1,9 @@ +# Behavior-preserving inference lint corrections + +Fresh PR #24 Clippy job 111191777174 reached the inference crate and reported 54 findings. This bounded slice addresses 36 annotation/style findings: 29 redundant method must_use attributes whose return types already carry that obligation; equivalent ordered filter/map and contains checks; Option early return via ?; derived SupportTier default with Unknown retained; and three borrowed Path comparisons instead of temporary PathBuf allocations. Public signatures, type-level must_use, conversion/error policy, and runtime identity semantics remain unchanged. + +The option regression now checks exact ordered unsupported diagnostics while a supported streaming option emits no issue. New tests retain Unknown's default serialized value and Path-component equality across inference/embedding/reranking matchers, including mismatched model rejection. Hosted checks explicitly run compatibility/model-contract suites and the matcher regression. Existing lifecycle builder/telemetry behavior is unchanged by removing redundant annotations. + +Formatting and whitespace pass locally. No local inference compilation or executed new Rust tests are claimed. Independent review and fresh hosted qualification remain pending. Nine high-arity functions, the public image-planning enum, and eight large-error returns are intentionally separate structural/API proposals, not suppressed or silently modified. + +Independent source review accepted tree cfc7168c7afd6904d00ea9818e67b8f3f17424cc, including diagnostic order/content, Unknown default, borrowed Path equivalence and retained type/Result must-use obligations. Fresh hosted tests and Clippy remain required. From 58df81689306debaced0e224e16576742376381d Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 04:37:00 -0700 Subject: [PATCH 04/25] fix(inference): box managed diagnostic and planned image payloads --- .github/workflows/quality-gates.yml | 6 +++++ CHANGELOG.md | 4 +++ .../pytorch_worker_image_contract_tests.rs | 2 +- crates/inference/src/gateway.rs | 4 +-- .../inference/src/image_generation_planner.rs | 6 ++--- .../src/image_generation_planner_tests.rs | 12 +++++++++ .../src/managed_runtime/contracts.rs | 27 ++++++++++++++++--- .../src/runtime_host_execution_port.rs | 2 +- .../2026-10-03-inference-payload-layout.md | 13 +++++++++ 9 files changed, 66 insertions(+), 10 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-inference-payload-layout.md diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml index 4aef59bdb..eceec701c 100644 --- a/.github/workflows/quality-gates.yml +++ b/.github/workflows/quality-gates.yml @@ -199,6 +199,12 @@ jobs: cargo test -p inference --lib model_contracts::tests cargo test -p inference --lib server::tests::runtime_matchers_preserve_path_component_comparison + - name: Run inference payload serialization and planner tests + run: | + cargo test -p inference --lib managed_runtime::contracts::tests + cargo test -p inference --lib image_generation_planner_tests + cargo test -p inference --lib pytorch_worker_image_contract_tests + - name: Run node-engine tests run: cargo test -p node-engine --lib diff --git a/CHANGELOG.md b/CHANGELOG.md index e293441d3..fa776e383 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,6 +14,10 @@ The format is based on Keep a Changelog. canonicalization records. ### Changed +- Rust source API: `ManagedRuntimeCommandResolutionError::MissingRuntimeVariant.diagnostic` + and `ImageGenerationPlanningOutcome::Planned.plan` now hold boxed payloads. + Direct constructors must box their values and consuming plan callers must unbox; + tagged JSON, diagnostic fields, and error Display text are unchanged. - Rust source API: `MediaConversionResult::try_new` now accepts one `MediaConversionResultInput` with the same eight named fields instead of eight positional arguments. Result fields and validation behavior are unchanged. diff --git a/crates/inference/src/backend/pytorch_worker_image_contract_tests.rs b/crates/inference/src/backend/pytorch_worker_image_contract_tests.rs index e46a74c4a..410d80dc2 100644 --- a/crates/inference/src/backend/pytorch_worker_image_contract_tests.rs +++ b/crates/inference/src/backend/pytorch_worker_image_contract_tests.rs @@ -262,7 +262,7 @@ fn test_pytorch_worker_generate_image_request_maps_from_validated_plan() { panic!("expected validated image plan"); }; - let worker_request = PyTorchGenerateImageRequest::from(&plan); + let worker_request = PyTorchGenerateImageRequest::from(plan.as_ref()); assert_eq!( worker_request.model_ref.model_id, diff --git a/crates/inference/src/gateway.rs b/crates/inference/src/gateway.rs index fea9f01a4..79333d5d3 100644 --- a/crates/inference/src/gateway.rs +++ b/crates/inference/src/gateway.rs @@ -1613,7 +1613,7 @@ impl InferenceGateway { reject_cancelled_execution_handle("image generation planning", &cancellation)?; match plan_image_generation_execution(input) { ImageGenerationPlanningOutcome::Planned { plan } => { - self.generate_image_from_plan_with_cancellation(plan, cancellation) + self.generate_image_from_plan_with_cancellation(*plan, cancellation) .await } ImageGenerationPlanningOutcome::Rejected { diagnostics } => { @@ -1699,7 +1699,7 @@ impl InferenceGateway { let execution_telemetry = self.start_execution_telemetry().await; let context = execution_telemetry.backend_execution_context(); let result = self - .generate_image_from_plan_with_context(plan, context) + .generate_image_from_plan_with_context(*plan, context) .await; let resource_observation = finish_execution_telemetry(execution_telemetry); record_planned_image_generation_lifecycle_result( diff --git a/crates/inference/src/image_generation_planner.rs b/crates/inference/src/image_generation_planner.rs index 7008fe941..0f9461cf0 100644 --- a/crates/inference/src/image_generation_planner.rs +++ b/crates/inference/src/image_generation_planner.rs @@ -142,7 +142,7 @@ pub enum ImageGenerationPlanningOutcome { /// The request has one canonical PyTorch/Diffusers execution plan. Planned { /// Validated execution plan. - plan: ImageGenerationExecutionPlan, + plan: Box, }, /// Planning failed closed with typed diagnostics. Rejected { @@ -493,7 +493,7 @@ pub fn plan_image_generation_execution( )]); }; ImageGenerationPlanningOutcome::Planned { - plan: ImageGenerationExecutionPlan { + plan: Box::new(ImageGenerationExecutionPlan { model_ref: input.package_facts.model_ref.clone(), artifact_entry_path, artifact_load_target: input.artifact_load_target.clone(), @@ -515,7 +515,7 @@ pub fn plan_image_generation_execution( denoising_scheduler, num_images_per_prompt: input.request.num_images_per_prompt, resource_estimates, - }, + }), } } diff --git a/crates/inference/src/image_generation_planner_tests.rs b/crates/inference/src/image_generation_planner_tests.rs index 7211a2b96..b754fc12c 100644 --- a/crates/inference/src/image_generation_planner_tests.rs +++ b/crates/inference/src/image_generation_planner_tests.rs @@ -126,9 +126,21 @@ fn planner_accepts_pumas_diffusers_stable_diffusion_facts() { backend_decision: &decision, }); + let encoded = serde_json::to_value(&outcome).expect("serialize planned outcome"); + let round_trip: ImageGenerationPlanningOutcome = + serde_json::from_value(encoded.clone()).expect("decode planned outcome"); + assert_eq!(round_trip, outcome); let ImageGenerationPlanningOutcome::Planned { plan } = outcome else { panic!("expected valid image-generation plan"); }; + assert_eq!( + encoded, + serde_json::json!({ "status": "planned", "plan": plan.as_ref() }) + ); + assert!( + std::mem::size_of::() + < std::mem::size_of::() + ); assert_eq!(plan.model_ref.model_id, "image/stable-diffusion/tiny-sd"); assert_eq!(plan.backend_id.as_str(), "pytorch"); diff --git a/crates/inference/src/managed_runtime/contracts.rs b/crates/inference/src/managed_runtime/contracts.rs index fac717e7b..5e13c0009 100644 --- a/crates/inference/src/managed_runtime/contracts.rs +++ b/crates/inference/src/managed_runtime/contracts.rs @@ -240,7 +240,7 @@ pub enum ManagedRuntimeCommandResolutionError { missing_file: String, }, MissingRuntimeVariant { - diagnostic: DeviceResolutionDiagnostic, + diagnostic: Box, requested_device: Option, missing_path: PathBuf, }, @@ -293,7 +293,7 @@ impl ManagedRuntimeCommandResolutionError { _ => None, }; Self::MissingRuntimeVariant { - diagnostic: DeviceResolutionDiagnostic { + diagnostic: Box::new(DeviceResolutionDiagnostic { code: DeviceResolutionDiagnosticCode::MissingRuntimeVariant, severity: DeviceResolutionDiagnosticSeverity::Error, message: format!( @@ -305,7 +305,7 @@ impl ManagedRuntimeCommandResolutionError { device_id: None, runtime_variant_id: Some(runtime_variant_id.clone()), backend_id: Some(backend_id), - }, + }), requested_device: None, missing_path, } @@ -370,6 +370,27 @@ mod tests { ); let encoded = serde_json::to_value(&error).expect("serialize command error"); + let message = "llama.cpp runtime variant 'llama_cpp.cuda' is selected but server binary is missing at /tmp/runtime/cuda/llama-server"; + let expected = serde_json::json!({ + "kind": "missing_runtime_variant", + "diagnostic": { + "code": "missing_runtime_variant", + "severity": "error", + "message": message, + "device_class": "cuda", + "runtime_variant_id": "llama_cpp.cuda", + "backend_id": "llama_cpp", + }, + "requested_device": null, + "missing_path": "/tmp/runtime/cuda/llama-server", + }); + assert_eq!(encoded, expected); + assert_eq!(error.to_string(), message); + let round_trip: ManagedRuntimeCommandResolutionError = + serde_json::from_value(expected).expect("existing error wire shape"); + assert_eq!(round_trip, error); + assert_eq!(round_trip.to_string(), message); + assert!(std::mem::size_of::() < 128); assert_eq!( encoded["kind"], diff --git a/crates/pantograph-embedded-runtime/src/runtime_host_execution_port.rs b/crates/pantograph-embedded-runtime/src/runtime_host_execution_port.rs index 7710dc220..306fb79a8 100644 --- a/crates/pantograph-embedded-runtime/src/runtime_host_execution_port.rs +++ b/crates/pantograph-embedded-runtime/src/runtime_host_execution_port.rs @@ -631,7 +631,7 @@ impl RuntimeHostBatchExecutionPort for EmbeddedRuntimeHostExecutionPort { }; let plan = match inference::plan_image_generation_execution(projection.planning_input()) { - ImageGenerationPlanningOutcome::Planned { plan } => plan, + ImageGenerationPlanningOutcome::Planned { plan } => *plan, ImageGenerationPlanningOutcome::Rejected { diagnostics } => { return Ok(rejected_batch_member_response( request, diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-inference-payload-layout.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-inference-payload-layout.md new file mode 100644 index 000000000..e331a69ba --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-inference-payload-layout.md @@ -0,0 +1,13 @@ +# Inference diagnostic and planning payload layout + +Fresh aggregate diagnostics identify two large payload owners: ManagedRuntimeCommandResolutionError::MissingRuntimeVariant embeds DeviceResolutionDiagnostic, and ImageGenerationPlanningOutcome::Planned embeds ImageGenerationExecutionPlan. Box only those payload fields. Leave all other error variants, request validation, planner decisions, facade context, registry resolution return types, and high-arity functions unchanged. + +Caller inventory: the managed-runtime diagnostic has one constructor and a Display arm reading its existing message; platform and serialization tests retain diagnostic access. The planning outcome has one constructor, two consuming gateway branches and one consuming embedded batch branch that unbox at their existing boundary, plus a worker projection test that explicitly borrows the raw plan. Inspection-only matches retain their behavior. + +These are explicit public Rust field-type changes, documented in the changelog despite inference being publish=false at workspace development version 0.1.0. Direct constructors must box payloads; owned plan consumers must unbox. Tagged JSON, diagnostic fields, Display messages, and underlying plan/result types remain unchanged. No release/version bump or unseen-consumer compatibility claim is made. + +The managed-runtime test checks exact full JSON, old-shape decoding, equality and exact Display text. The successful planner fixture retains its detailed plan assertions and adds exact outer tagged shape/round-trip/layout checks. Hosted steps run managed-runtime contracts, all existing planner acceptance/rejection tests, and worker image projection tests. The inline size assertions are not execution-performance measurements. + +Formatting and whitespace pass locally. No local workspace compilation/new Rust execution is claimed. Independent source review and fresh hosted inference/embedded/workflow qualification are pending. + +Independent source review accepted tree d8784bd001db2ffef90b0c97465b167bf1c51472, including the bounded two-field layout change, consuming caller adaptations, exact JSON/Display assertions and retained planner suites. Fresh hosted execution is still required. From 1cae42ec9ff5a904ffe780a2ae129c2c3975fcdf Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 04:47:05 -0700 Subject: [PATCH 05/25] fix(ci): execute feature-gated PyTorch projection tests --- .github/workflows/quality-gates.yml | 4 +++- .../reports/2026-10-03-inference-payload-layout.md | 7 +++++++ 2 files changed, 10 insertions(+), 1 deletion(-) diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml index eceec701c..65bde6762 100644 --- a/.github/workflows/quality-gates.yml +++ b/.github/workflows/quality-gates.yml @@ -203,7 +203,9 @@ jobs: run: | cargo test -p inference --lib managed_runtime::contracts::tests cargo test -p inference --lib image_generation_planner_tests - cargo test -p inference --lib pytorch_worker_image_contract_tests + cargo test -p inference --features backend-pytorch --lib pytorch_worker_image_contract_tests -- --list > "$RUNNER_TEMP/pytorch-image-contract-tests.list" + python3 -c 'import pathlib, sys; names = [line for line in pathlib.Path(sys.argv[1]).read_text().splitlines() if line.startswith("backend::pytorch::pytorch_worker_image_contract_tests::") and line.endswith(": test")]; print("PyTorch image contract tests discovered:", len(names)); sys.exit(0 if names else 1)' "$RUNNER_TEMP/pytorch-image-contract-tests.list" + cargo test -p inference --features backend-pytorch --lib pytorch_worker_image_contract_tests - name: Run node-engine tests run: cargo test -p node-engine --lib diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-inference-payload-layout.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-inference-payload-layout.md index e331a69ba..24f2eee87 100644 --- a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-inference-payload-layout.md +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-inference-payload-layout.md @@ -11,3 +11,10 @@ The managed-runtime test checks exact full JSON, old-shape decoding, equality an Formatting and whitespace pass locally. No local workspace compilation/new Rust execution is claimed. Independent source review and fresh hosted inference/embedded/workflow qualification are pending. Independent source review accepted tree d8784bd001db2ffef90b0c97465b167bf1c51472, including the bounded two-field layout change, consuming caller adaptations, exact JSON/Display assertions and retained planner suites. Fresh hosted execution is still required. + + +## Hosted qualification correction + +At head 58df81689306debaced0e224e16576742376381d, the exact managed-runtime error test and all 25 planner tests pass. The worker projection command ran zero tests because inference's default feature is backend-llamacpp. The corrected command explicitly enables backend-pytorch, checks that the filtered suite lists at least one test, and then runs that same feature/filter selection. Worker projection execution remains unqualified until the corrected hosted step succeeds. Aggregate inference Clippy is down to 11 remaining findings; those are separate scope. + +Independent source review accepted correction tree 6e3942251a33f311f829aba75b10859bb46482f6: explicit feature selection and nonzero exact-module discovery close the false-green gap. The final receipt must report the actual executed count; the prior zero-test step remains unqualified. From 4bbf55eda8c095914b5288a59a47765a540fcf39 Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 04:49:14 -0700 Subject: [PATCH 06/25] refactor(inference): group private gateway lifecycle contexts --- .github/workflows/quality-gates.yml | 6 + crates/inference/src/gateway.rs | 334 ++++++++++-------- crates/inference/src/gateway_tests.rs | 79 +++++ .../2026-10-03-gateway-private-contexts.md | 11 + 4 files changed, 279 insertions(+), 151 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-gateway-private-contexts.md diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml index 65bde6762..4d6123532 100644 --- a/.github/workflows/quality-gates.yml +++ b/.github/workflows/quality-gates.yml @@ -207,6 +207,12 @@ jobs: python3 -c 'import pathlib, sys; names = [line for line in pathlib.Path(sys.argv[1]).read_text().splitlines() if line.startswith("backend::pytorch::pytorch_worker_image_contract_tests::") and line.endswith(": test")]; print("PyTorch image contract tests discovered:", len(names)); sys.exit(0 if names else 1)' "$RUNNER_TEMP/pytorch-image-contract-tests.list" cargo test -p inference --features backend-pytorch --lib pytorch_worker_image_contract_tests + - name: Run gateway attribution and warmup regression tests + run: | + cargo test -p inference --lib gateway::tests::private_lifecycle_context + cargo test -p inference --lib gateway::tests::test_runtime_lifecycle_snapshot + cargo test -p inference --lib gateway::tests::test_failed_restart_clears_active_runtime_config_and_attempted_modes + - name: Run node-engine tests run: cargo test -p node-engine --lib diff --git a/crates/inference/src/gateway.rs b/crates/inference/src/gateway.rs index 79333d5d3..540d1fea3 100644 --- a/crates/inference/src/gateway.rs +++ b/crates/inference/src/gateway.rs @@ -46,9 +46,9 @@ use crate::types::{ InferenceCompatibilityIssueSummary, InferenceCompatibilityReportSummary, InferenceEmbeddingResult, InferenceExecutionInput, InferenceExecutionRequest, InferenceExecutionRequestValidationError, InferenceExecutionResult, - InferenceRequestLifecycleEvent, InferenceRequestLifecycleEventKind, - InferenceRequestLifecycleEventSink, InferenceUsage, RerankRequest, RerankResponse, - RuntimeLifecycleSnapshot, ServerModeInfo, + InferenceRequestLifecycleEvent, InferenceRequestLifecycleEventContext, + InferenceRequestLifecycleEventKind, InferenceRequestLifecycleEventSink, InferenceUsage, + RerankRequest, RerankResponse, RuntimeLifecycleSnapshot, ServerModeInfo, }; use crate::{ BackendExecutionContext, InferenceExecutionCancellationHandle, @@ -155,6 +155,15 @@ pub struct InferenceGateway { runtime_instance_sequence: Arc, } +struct RuntimeWarmupStartContext<'a> { + config: &'a BackendConfig, + previous_last_inference_config: Option, + previous_runtime_instance_id: Option, + runtime_id: String, + warmup_started_at_ms: u64, + warmup_timing_attempt_id: pantograph_timing_contracts::WorkflowTimingAttemptId, +} + fn config_model_target(config: &BackendConfig) -> Option { config .model_path @@ -657,12 +666,14 @@ impl InferenceGateway { } self.record_start_result( - config, - previous_last_inference_config, - previous_runtime_instance_id, - runtime_id, - warmup_started_at_ms, - warmup_timing_attempt_id, + RuntimeWarmupStartContext { + config, + previous_last_inference_config, + previous_runtime_instance_id, + runtime_id, + warmup_started_at_ms, + warmup_timing_attempt_id, + }, start_result, ) .await @@ -670,14 +681,17 @@ impl InferenceGateway { async fn record_start_result( &self, - config: &BackendConfig, - previous_last_inference_config: Option, - previous_runtime_instance_id: Option, - runtime_id: String, - warmup_started_at_ms: u64, - warmup_timing_attempt_id: pantograph_timing_contracts::WorkflowTimingAttemptId, + context: RuntimeWarmupStartContext<'_>, start_result: Result, ) -> Result<(), GatewayError> { + let RuntimeWarmupStartContext { + config, + previous_last_inference_config, + previous_runtime_instance_id, + runtime_id, + warmup_started_at_ms, + warmup_timing_attempt_id, + } = context; match start_result { Ok(start_outcome) => { let mut current_runtime_config = self.current_runtime_config.write().await; @@ -1014,14 +1028,17 @@ impl InferenceGateway { record_inference_lifecycle_event( lifecycle_sink.as_ref(), - request_id.clone(), - task_id.clone(), - backend_key.clone(), - runtime_id.clone(), - runtime_instance_id.clone(), - selected_device_class, - selected_device_id.clone(), - model_id.clone(), + InferenceRequestLifecycleEventContext { + request_id: request_id.clone(), + task_id: task_id.clone(), + backend_key: backend_key.clone(), + runtime_id: runtime_id.clone(), + runtime_instance_id: runtime_instance_id.clone(), + selected_device_class, + selected_device_id: selected_device_id.clone(), + model_id: model_id.clone(), + ..Default::default() + }, InferenceRequestLifecycleEventKind::Started, None, ); @@ -1065,14 +1082,17 @@ impl InferenceGateway { ); record_inference_lifecycle_event( lifecycle_sink.as_ref(), - request_id, - task_id, - backend_key, - runtime_id, - runtime_instance_id, - selected_device_class, - selected_device_id, - model_id, + InferenceRequestLifecycleEventContext { + request_id, + task_id, + backend_key, + runtime_id, + runtime_instance_id, + selected_device_class, + selected_device_id, + model_id, + ..Default::default() + }, InferenceRequestLifecycleEventKind::CleanupCompleted, None, ); @@ -1219,14 +1239,17 @@ impl InferenceGateway { record_inference_lifecycle_phase_event( lifecycle_sink.as_ref(), InferenceLifecyclePhase::TaskValidation, - request_id.clone(), - task_id.clone(), - backend_key.clone(), - runtime_id.clone(), - runtime_instance_id.clone(), - selected_device_class, - selected_device_id.clone(), - model_id.clone(), + InferenceRequestLifecycleEventContext { + request_id: request_id.clone(), + task_id: task_id.clone(), + backend_key: backend_key.clone(), + runtime_id: runtime_id.clone(), + runtime_instance_id: runtime_instance_id.clone(), + selected_device_class, + selected_device_id: selected_device_id.clone(), + model_id: model_id.clone(), + ..Default::default() + }, InferenceRequestLifecycleEventKind::Started, None, ); @@ -1279,14 +1302,17 @@ impl InferenceGateway { record_inference_lifecycle_phase_event( lifecycle_sink.as_ref(), InferenceLifecyclePhase::Preprocessing, - request_id.clone(), - task_id.clone(), - backend_key.clone(), - runtime_id.clone(), - runtime_instance_id.clone(), - selected_device_class, - selected_device_id.clone(), - model_id.clone(), + InferenceRequestLifecycleEventContext { + request_id: request_id.clone(), + task_id: task_id.clone(), + backend_key: backend_key.clone(), + runtime_id: runtime_id.clone(), + runtime_instance_id: runtime_instance_id.clone(), + selected_device_class, + selected_device_id: selected_device_id.clone(), + model_id: model_id.clone(), + ..Default::default() + }, InferenceRequestLifecycleEventKind::Started, None, ); @@ -1352,14 +1378,17 @@ impl InferenceGateway { let model_id = non_empty_model_id(model); record_inference_lifecycle_event( lifecycle_sink.as_ref(), - request_id.clone(), - Some("embedding".to_string()), - backend_key.clone(), - runtime_id.clone(), - runtime_instance_id.clone(), - selected_device_class, - selected_device_id.clone(), - model_id.clone(), + InferenceRequestLifecycleEventContext { + request_id: request_id.clone(), + task_id: Some("embedding".to_string()), + backend_key: backend_key.clone(), + runtime_id: runtime_id.clone(), + runtime_instance_id: runtime_instance_id.clone(), + selected_device_class, + selected_device_id: selected_device_id.clone(), + model_id: model_id.clone(), + ..Default::default() + }, InferenceRequestLifecycleEventKind::Started, None, ); @@ -1444,14 +1473,17 @@ impl InferenceGateway { let model_id = non_empty_model_id(&request.model); record_inference_lifecycle_event( lifecycle_sink.as_ref(), - request_id.clone(), - Some("rerank".to_string()), - backend_key.clone(), - runtime_id.clone(), - runtime_instance_id.clone(), - selected_device_class, - selected_device_id.clone(), - model_id.clone(), + InferenceRequestLifecycleEventContext { + request_id: request_id.clone(), + task_id: Some("rerank".to_string()), + backend_key: backend_key.clone(), + runtime_id: runtime_id.clone(), + runtime_instance_id: runtime_instance_id.clone(), + selected_device_class, + selected_device_id: selected_device_id.clone(), + model_id: model_id.clone(), + ..Default::default() + }, InferenceRequestLifecycleEventKind::Started, None, ); @@ -1781,14 +1813,17 @@ impl InferenceGateway { let model_id = non_empty_model_id(&request.model); record_inference_lifecycle_event( lifecycle_sink.as_ref(), - request_id.clone(), - Some("image_generation".to_string()), - backend_key.clone(), - runtime_id.clone(), - runtime_instance_id.clone(), - selected_device_class, - selected_device_id.clone(), - model_id.clone(), + InferenceRequestLifecycleEventContext { + request_id: request_id.clone(), + task_id: Some("image_generation".to_string()), + backend_key: backend_key.clone(), + runtime_id: runtime_id.clone(), + runtime_instance_id: runtime_instance_id.clone(), + selected_device_class, + selected_device_id: selected_device_id.clone(), + model_id: model_id.clone(), + ..Default::default() + }, InferenceRequestLifecycleEventKind::Started, None, ); @@ -1865,14 +1900,17 @@ impl InferenceGateway { record_inference_lifecycle_phase_event( lifecycle_sink.as_ref(), InferenceLifecyclePhase::TaskValidation, - request_id.clone(), - task_id.clone(), - backend_key.clone(), - runtime_id.clone(), - runtime_instance_id.clone(), - selected_device_class, - selected_device_id.clone(), - model_id.clone(), + InferenceRequestLifecycleEventContext { + request_id: request_id.clone(), + task_id: task_id.clone(), + backend_key: backend_key.clone(), + runtime_id: runtime_id.clone(), + runtime_instance_id: runtime_instance_id.clone(), + selected_device_class, + selected_device_id: selected_device_id.clone(), + model_id: model_id.clone(), + ..Default::default() + }, InferenceRequestLifecycleEventKind::Started, None, ); @@ -1948,14 +1986,17 @@ impl InferenceGateway { } record_inference_lifecycle_event( lifecycle_sink.as_ref(), - request_id.clone(), - task_id.clone(), - backend_key.clone(), - runtime_id.clone(), - runtime_instance_id.clone(), - selected_device_class, - selected_device_id.clone(), - model_id.clone(), + InferenceRequestLifecycleEventContext { + request_id: request_id.clone(), + task_id: task_id.clone(), + backend_key: backend_key.clone(), + runtime_id: runtime_id.clone(), + runtime_instance_id: runtime_instance_id.clone(), + selected_device_class, + selected_device_id: selected_device_id.clone(), + model_id: model_id.clone(), + ..Default::default() + }, InferenceRequestLifecycleEventKind::Started, None, ); @@ -2307,14 +2348,17 @@ impl LifecycleStream { fn record(&self, kind: InferenceRequestLifecycleEventKind, detail: Option) { record_inference_lifecycle_event( self.lifecycle_sink.as_ref(), - self.request_id.clone(), - self.task_id.clone(), - self.backend_key.clone(), - self.runtime_id.clone(), - self.runtime_instance_id.clone(), - self.selected_device_class, - self.selected_device_id.clone(), - self.model_id.clone(), + InferenceRequestLifecycleEventContext { + request_id: self.request_id.clone(), + task_id: self.task_id.clone(), + backend_key: self.backend_key.clone(), + runtime_id: self.runtime_id.clone(), + runtime_instance_id: self.runtime_instance_id.clone(), + selected_device_class: self.selected_device_class, + selected_device_id: self.selected_device_id.clone(), + model_id: self.model_id.clone(), + ..Default::default() + }, kind, detail, ); @@ -3118,28 +3162,14 @@ fn typed_text_generation_stream_request_json( fn record_inference_lifecycle_event( sink: &dyn InferenceRequestLifecycleEventSink, - request_id: Option, - task_id: Option, - backend_key: Option, - runtime_id: Option, - runtime_instance_id: Option, - selected_device_class: Option, - selected_device_id: Option, - model_id: Option, + context: InferenceRequestLifecycleEventContext, kind: InferenceRequestLifecycleEventKind, detail: Option, ) { record_inference_lifecycle_phase_event( sink, InferenceLifecyclePhase::BackendExecution, - request_id, - task_id, - backend_key, - runtime_id, - runtime_instance_id, - selected_device_class, - selected_device_id, - model_id, + context, kind, detail, ); @@ -3148,28 +3178,21 @@ fn record_inference_lifecycle_event( fn record_inference_lifecycle_phase_event( sink: &dyn InferenceRequestLifecycleEventSink, phase: InferenceLifecyclePhase, - request_id: Option, - task_id: Option, - backend_key: Option, - runtime_id: Option, - runtime_instance_id: Option, - selected_device_class: Option, - selected_device_id: Option, - model_id: Option, + context: InferenceRequestLifecycleEventContext, kind: InferenceRequestLifecycleEventKind, detail: Option, ) { record_inference_lifecycle_phase_event_with_option_diagnostics( sink, phase, - request_id, - task_id, - backend_key, - runtime_id, - runtime_instance_id, - selected_device_class, - selected_device_id, - model_id, + context.request_id, + context.task_id, + context.backend_key, + context.runtime_id, + context.runtime_instance_id, + context.selected_device_class, + context.selected_device_id, + context.model_id, kind, detail, Vec::new(), @@ -3778,14 +3801,17 @@ fn record_typed_lifecycle_result_with_option_diagnostics( record_inference_lifecycle_phase_event( sink, InferenceLifecyclePhase::BackendExecution, - request_id, - task_id, - backend_key, - runtime_id, - runtime_instance_id, - selected_device_class, - selected_device_id, - model_id, + InferenceRequestLifecycleEventContext { + request_id, + task_id, + backend_key, + runtime_id, + runtime_instance_id, + selected_device_class, + selected_device_id, + model_id, + ..Default::default() + }, InferenceRequestLifecycleEventKind::CleanupCompleted, None, ); @@ -3867,14 +3893,17 @@ fn record_successful_non_streaming_lifecycle_phase( record_inference_lifecycle_phase_event( sink, phase.clone(), - request_id.clone(), - task_id.clone(), - backend_key.clone(), - runtime_id.clone(), - runtime_instance_id.clone(), - selected_device_class, - selected_device_id.clone(), - model_id.clone(), + InferenceRequestLifecycleEventContext { + request_id: request_id.clone(), + task_id: task_id.clone(), + backend_key: backend_key.clone(), + runtime_id: runtime_id.clone(), + runtime_instance_id: runtime_instance_id.clone(), + selected_device_class, + selected_device_id: selected_device_id.clone(), + model_id: model_id.clone(), + ..Default::default() + }, InferenceRequestLifecycleEventKind::Started, None, ); @@ -4037,14 +4066,17 @@ fn record_non_streaming_lifecycle_phase_result_with_references( record_inference_lifecycle_phase_event( sink, phase, - request_id, - task_id, - backend_key, - runtime_id, - runtime_instance_id, - selected_device_class, - selected_device_id, - model_id, + InferenceRequestLifecycleEventContext { + request_id, + task_id, + backend_key, + runtime_id, + runtime_instance_id, + selected_device_class, + selected_device_id, + model_id, + ..Default::default() + }, InferenceRequestLifecycleEventKind::CleanupCompleted, None, ); diff --git a/crates/inference/src/gateway_tests.rs b/crates/inference/src/gateway_tests.rs index 91e5d298a..094354ade 100644 --- a/crates/inference/src/gateway_tests.rs +++ b/crates/inference/src/gateway_tests.rs @@ -5071,3 +5071,82 @@ async fn selected_text_invalid_handoffs_have_no_backend_effects() { .is_err()); assert!(effects.lock().unwrap().is_empty()); } + +#[test] +fn private_lifecycle_context_preserves_complete_attribution_and_absence() { + let sink = RecordingLifecycleSink::default(); + let device_id = InferenceDeviceId::parse("cuda:0").expect("device id"); + let context = InferenceRequestLifecycleEventContext { + request_id: Some("request-1".to_string()), + task_id: Some("text_generation".to_string()), + backend_key: Some("llama_cpp".to_string()), + runtime_id: Some("llama_cpp.cuda".to_string()), + runtime_instance_id: Some("instance-1".to_string()), + selected_device_class: Some(InferenceDeviceClass::Cuda), + selected_device_id: Some(device_id.clone()), + model_id: Some("pumas://models/example".to_string()), + ..Default::default() + }; + record_inference_lifecycle_event( + &sink, + context.clone(), + InferenceRequestLifecycleEventKind::Started, + Some("start detail".to_string()), + ); + record_inference_lifecycle_phase_event( + &sink, + InferenceLifecyclePhase::Preprocessing, + context, + InferenceRequestLifecycleEventKind::Completed, + None, + ); + record_inference_lifecycle_event( + &sink, + InferenceRequestLifecycleEventContext::default(), + InferenceRequestLifecycleEventKind::CleanupCompleted, + None, + ); + let events = sink.events(); + assert_eq!(events.len(), 3); + for (index, (phase, kind, detail)) in [ + ( + InferenceLifecyclePhase::BackendExecution, + InferenceRequestLifecycleEventKind::Started, + Some("start detail".to_string()), + ), + ( + InferenceLifecyclePhase::Preprocessing, + InferenceRequestLifecycleEventKind::Completed, + None, + ), + ] + .into_iter() + .enumerate() + { + let expected = + InferenceRequestLifecycleEvent::builder(phase, kind, events[index].occurred_at_ms) + .with_request_id(Some("request-1".to_string())) + .with_task_id(Some("text_generation".to_string())) + .with_backend_key(Some("llama_cpp".to_string())) + .with_runtime_id(Some("llama_cpp.cuda".to_string())) + .with_runtime_instance_id(Some("instance-1".to_string())) + .with_selected_device_class(Some(InferenceDeviceClass::Cuda)) + .with_selected_device_id(Some(device_id.clone())) + .with_model_id(Some("pumas://models/example".to_string())) + .with_detail(detail) + .build(); + assert_eq!(events[index], expected); + assert_eq!(events[index].selected_runtime_variant_id, None); + assert_eq!(events[index].selected_network_node_id, None); + assert_eq!(events[index].resolved_artifact_kind, None); + } + assert_eq!( + events[2], + InferenceRequestLifecycleEvent::builder( + InferenceLifecyclePhase::BackendExecution, + InferenceRequestLifecycleEventKind::CleanupCompleted, + events[2].occurred_at_ms, + ) + .build() + ); +} diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-gateway-private-contexts.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-gateway-private-contexts.md new file mode 100644 index 000000000..c577ea8e6 --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-gateway-private-contexts.md @@ -0,0 +1,11 @@ +# Private gateway lifecycle and warmup context grouping + +Replace eight repeated attribution parameters in two private lifecycle helpers with the existing InferenceRequestLifecycleEventContext. All thirteen existing call sites populate the same eight values; previously absent fields remain None via Default. The lower diagnostic recorder is unchanged. The context is moved through the helper, avoiding extra attribution clones. + +Group the one-caller private record_start_result inputs into RuntimeWarmupStartContext while leaving its Result separate. Destructuring restores the same local names before the byte-identical success/rollback/timing body. No public generation/server signature, lock scope, timestamp source, runtime identity, lifecycle selection, or error behavior changes. + +A complete-event regression checks both helper routes, all attribution values, detail, phase/kind, and explicit absence of variant/network/artifact fields; a default-context event confirms no invented attribution. Hosted steps also run existing warmup success, normalized failure and failed-restart rollback tests. + +Formatting/whitespace checks and a source comparison of the unchanged warmup behavior block pass locally. No local Rust execution is claimed. Independent source review and fresh hosted tests/Clippy remain pending. Existing unrelated high-arity helpers and public APIs are untouched; no new lint suppression is added. + +Independent source review accepted tree 03e481cd24fd8765c6b11c865ec73ff8619dd1e8. The reviewed PR #26 feature-gated projection-test correction was merged cleanly as the publication base; gateway source/test contents are unchanged by that inheritance. Actual hosted lifecycle/warmup execution remains required. From 23c0c141c4620e3f2d6f462648098f399689b9b0 Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 05:00:44 -0700 Subject: [PATCH 07/25] fix(inference): box two concrete error return boundaries --- .github/workflows/quality-gates.yml | 5 ++ CHANGELOG.md | 4 + crates/inference/src/managed_binaries.rs | 74 +++++++++++++++++-- crates/inference/src/model_contracts.rs | 23 +++--- crates/inference/tests/model_contracts.rs | 28 ++++++- .../2026-10-03-inference-error-boundaries.md | 11 +++ 6 files changed, 124 insertions(+), 21 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-inference-error-boundaries.md diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml index 4d6123532..548da9f1e 100644 --- a/.github/workflows/quality-gates.yml +++ b/.github/workflows/quality-gates.yml @@ -213,6 +213,11 @@ jobs: cargo test -p inference --lib gateway::tests::test_runtime_lifecycle_snapshot cargo test -p inference --lib gateway::tests::test_failed_restart_clears_active_runtime_config_and_attempted_modes + - name: Run typed error boundary regression tests + run: | + cargo test -p inference --lib managed_binaries::tests + cargo test -p inference --test model_contracts task_registry_resolution_ + - name: Run node-engine tests run: cargo test -p node-engine --lib diff --git a/CHANGELOG.md b/CHANGELOG.md index fa776e383..d74a50be3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,6 +14,10 @@ The format is based on Keep a Changelog. canonicalization records. ### Changed +- Rust source API: `resolve_managed_binary_command` and + `resolve_task_registry_entry_from_evidence` now return boxed concrete errors. + Error variants, diagnostic JSON and Display text remain unchanged; callers + matching an owned error must dereference the box. - Rust source API: `ManagedRuntimeCommandResolutionError::MissingRuntimeVariant.diagnostic` and `ImageGenerationPlanningOutcome::Planned.plan` now hold boxed payloads. Direct constructors must box their values and consuming plan callers must unbox; diff --git a/crates/inference/src/managed_binaries.rs b/crates/inference/src/managed_binaries.rs index 6dae960d5..81fa2a374 100644 --- a/crates/inference/src/managed_binaries.rs +++ b/crates/inference/src/managed_binaries.rs @@ -112,13 +112,13 @@ pub fn resolve_managed_binary_command( app_data_dir: &Path, id: ManagedBinaryId, args: &[&str], -) -> Result { +) -> Result> { let snapshot = managed_runtime_snapshot(app_data_dir, id) .map_err(ManagedBinaryFacadeError::RuntimeStatus)?; if !snapshot.available || snapshot.readiness_state != ManagedRuntimeReadinessState::Ready { let install_root = selected_install_root(&snapshot); - return Err(ManagedBinaryFacadeError::RuntimeNotReady { + return Err(Box::new(ManagedBinaryFacadeError::RuntimeNotReady { key: ManagedBinaryKey::runtime(snapshot.id), display_name: snapshot.display_name, readiness_state: snapshot.readiness_state, @@ -126,7 +126,7 @@ pub fn resolve_managed_binary_command( install_root, missing_files: snapshot.missing_files, unavailable_reason: snapshot.unavailable_reason, - }); + })); } let key = managed_runtime_dependency_key(id).ok_or_else(|| { @@ -140,13 +140,13 @@ pub fn resolve_managed_binary_command( .map(resolved_command_from_dependency_command) .map_err(|source| { let install_root = selected_install_root(&snapshot); - ManagedBinaryFacadeError::RuntimeCommandResolution { + Box::new(ManagedBinaryFacadeError::RuntimeCommandResolution { key: ManagedBinaryKey::runtime(snapshot.id), display_name: snapshot.display_name, selected_version: snapshot.selection.selected_version, install_root, source, - } + }) }) } @@ -196,14 +196,14 @@ mod tests { fn resolve_command_reports_facade_not_ready_context() { let temp = tempfile::tempdir().expect("temp dir"); - let error = resolve_managed_binary_command( + let error: Box = resolve_managed_binary_command( temp.path(), ManagedBinaryId::LlamaCpp, &["--port", "0"], ) .expect_err("missing llama.cpp should fail before command resolution"); - match error { + match *error { ManagedBinaryFacadeError::RuntimeNotReady { key, readiness_state, @@ -217,4 +217,64 @@ mod tests { other => panic!("unexpected error: {other}"), } } + + #[test] + fn boxed_facade_errors_preserve_variants_and_exact_display() { + let not_ready = Box::new(ManagedBinaryFacadeError::RuntimeNotReady { + key: ManagedBinaryKey::runtime(ManagedBinaryId::LlamaCpp), + display_name: "llama.cpp".to_string(), + readiness_state: ManagedRuntimeReadinessState::Missing, + selected_version: Some("v1".to_string()), + install_root: Some("/runtime".to_string()), + missing_files: vec!["server".to_string(), "library".to_string()], + unavailable_reason: Some("not installed".to_string()), + }); + assert_eq!(not_ready.to_string(), "llama.cpp is not ready for launch (Missing, selected version v1, install root /runtime, missing server, library: not installed)"); + assert!(matches!( + *not_ready, + ManagedBinaryFacadeError::RuntimeNotReady { .. } + )); + let command = Box::new(ManagedBinaryFacadeError::RuntimeCommandResolution { + key: ManagedBinaryKey::runtime(ManagedBinaryId::LlamaCpp), + display_name: "llama.cpp".to_string(), + selected_version: Some("v1".to_string()), + install_root: Some("/runtime".to_string()), + source: "denied".to_string(), + }); + assert_eq!(command.to_string(), "failed to resolve llama.cpp launch command for selected version v1 at /runtime: denied"); + assert!(matches!( + *command, + ManagedBinaryFacadeError::RuntimeCommandResolution { .. } + )); + let status = Box::new(ManagedBinaryFacadeError::RuntimeStatus( + "unreadable".to_string(), + )); + assert_eq!( + status.to_string(), + "failed to read managed runtime status: unreadable" + ); + } + + #[test] + fn successful_command_projection_preserves_launch_fields_and_argument_boundaries() { + let command = resolved_command_from_dependency_command(ResolvedManagedDependencyCommand { + key: ManagedDependencyKey::RuntimeSidecar(RuntimeSidecarDependencyId::LlamaCpp), + executable_path: "/runtime/server".to_string(), + working_directory: "/runtime".to_string(), + args: vec!["--model".to_string(), "models/a b.gguf".to_string()], + env_overrides: vec![("RUNTIME_MODE".to_string(), "test".to_string())], + pid_file: Some("/runtime/server.pid".to_string()), + }); + assert_eq!(command.executable_path, PathBuf::from("/runtime/server")); + assert_eq!(command.working_directory, PathBuf::from("/runtime")); + assert_eq!( + command.args, + vec![OsString::from("--model"), OsString::from("models/a b.gguf")] + ); + assert_eq!( + command.env_overrides, + vec![(OsString::from("RUNTIME_MODE"), OsString::from("test"))] + ); + assert_eq!(command.pid_file, Some(PathBuf::from("/runtime/server.pid"))); + } } diff --git a/crates/inference/src/model_contracts.rs b/crates/inference/src/model_contracts.rs index c88a1e4cb..03b125c54 100644 --- a/crates/inference/src/model_contracts.rs +++ b/crates/inference/src/model_contracts.rs @@ -800,10 +800,12 @@ pub fn resolve_task_registry_entry(value: &str) -> Option { /// task signature. pub fn resolve_task_registry_entry_from_evidence( evidence: &TaskEvidence, -) -> Result { +) -> Result> { let labels = task_evidence_labels(evidence); if labels.is_empty() { - return Err(TaskRegistryResolutionDiagnostic::missing_task_evidence()); + return Err(Box::new( + TaskRegistryResolutionDiagnostic::missing_task_evidence(), + )); } let mut resolved_entries = Vec::new(); @@ -814,8 +816,8 @@ pub fn resolve_task_registry_entry_from_evidence( } let Some(first) = resolved_entries.first().cloned() else { - return Err(TaskRegistryResolutionDiagnostic::unsupported_task_label( - labels, + return Err(Box::new( + TaskRegistryResolutionDiagnostic::unsupported_task_label(labels), )); }; @@ -830,21 +832,20 @@ pub fn resolve_task_registry_entry_from_evidence( } if canonical_task_ids.len() > 1 { - return Err(TaskRegistryResolutionDiagnostic::conflicting_task_evidence( - labels, - canonical_task_ids, + return Err(Box::new( + TaskRegistryResolutionDiagnostic::conflicting_task_evidence(labels, canonical_task_ids), )); } if !first.matches_task_evidence(evidence) { - return Err(TaskRegistryResolutionDiagnostic::unsupported_task_label( - labels, + return Err(Box::new( + TaskRegistryResolutionDiagnostic::unsupported_task_label(labels), )); } if !first.matches_modality_evidence(evidence) { - return Err(TaskRegistryResolutionDiagnostic::modality_mismatch( - &first, evidence, + return Err(Box::new( + TaskRegistryResolutionDiagnostic::modality_mismatch(&first, evidence), )); } diff --git a/crates/inference/tests/model_contracts.rs b/crates/inference/tests/model_contracts.rs index a8412597b..87340781f 100644 --- a/crates/inference/tests/model_contracts.rs +++ b/crates/inference/tests/model_contracts.rs @@ -19,8 +19,8 @@ use inference::{ PackageSizeRole, ProcessorComponentKind, PumasArtifactLoadPathKind, PumasArtifactLoadTarget, PumasModelRef, ResolvedModelPackageFacts, ResolvedModelSource, ResolvedModelSourceKind, RuntimeLifecycleSnapshot, SupportTier, TaskEvidence, TaskExecutionBehavior, TaskFamily, - TaskRegistryEntry, TaskRegistryResolutionDiagnosticKind, TaskRequestContract, - TaskStreamingSupport, MODEL_PACKAGE_FACTS_CONTRACT_VERSION, + TaskRegistryEntry, TaskRegistryResolutionDiagnostic, TaskRegistryResolutionDiagnosticKind, + TaskRequestContract, TaskStreamingSupport, MODEL_PACKAGE_FACTS_CONTRACT_VERSION, }; const PACKAGE_FACT_FIXTURES: &[(&str, &str)] = &[ @@ -1156,6 +1156,11 @@ fn task_registry_resolution_returns_validated_entry_from_package_evidence() { .expect("text generation evidence should resolve"); assert_eq!(entry.task_id, InferenceTaskId::TextGeneration); + let expected = resolve_task_registry_entry("text_generation").expect("canonical entry"); + assert_eq!( + serde_json::to_value(&entry).expect("resolved entry"), + serde_json::to_value(expected).expect("canonical entry") + ); } #[test] @@ -1181,7 +1186,17 @@ fn task_registry_resolution_reports_unsupported_task_evidence() { ); let encoded = serde_json::to_value(&diagnostic).expect("encode diagnostic"); - assert_eq!(encoded["kind"], serde_json::json!("unsupported_task_label")); + assert_eq!( + encoded, + serde_json::json!({ + "kind": "unsupported_task_label", + "message": "package task evidence does not match a canonical inference task registry entry", + "labels": ["object_detection", "object-detection"], + }) + ); + let decoded: TaskRegistryResolutionDiagnostic = + serde_json::from_value(encoded).expect("decode existing diagnostic shape"); + assert_eq!(&decoded, diagnostic.as_ref()); } #[test] @@ -1198,6 +1213,13 @@ fn task_registry_resolution_reports_missing_task_evidence() { TaskRegistryResolutionDiagnosticKind::MissingTaskEvidence ); assert!(diagnostic.labels.is_empty()); + assert_eq!( + serde_json::to_value(&diagnostic).expect("missing diagnostic"), + serde_json::json!({ + "kind": "missing_task_evidence", + "message": "package task evidence does not include a task label", + }) + ); } #[test] diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-inference-error-boundaries.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-inference-error-boundaries.md new file mode 100644 index 000000000..f4b2559d7 --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-inference-error-boundaries.md @@ -0,0 +1,11 @@ +# Two typed error return boundaries + +Fresh inference diagnostics identify only two remaining oversized error return owners after the managed diagnostic payload repair. Return Box from resolve_managed_binary_command and Box from resolve_task_registry_entry_from_evidence. Keep the concrete error enums/DTOs, all fields, Display/Debug messages, diagnostic JSON, successful result types, and validation/branch order unchanged. No other Result is boxed. + +This explicitly changes two public Rust return types and adds allocation on their failure paths. Direct owned pattern matching needs dereferencing; the existing facade test is migrated. Repository consumers otherwise inspect fields, format diagnostics, or discard errors and retain that behavior. The changelog records the source change despite inference being publish=false; unseen consumers are not claimed compatible. + +Tests cover actual missing-runtime facade resolution, all facade variants/exact Display, successful command field/argument-boundary projection, full successful canonical registry output, and exact diagnostic JSON/raw-shape decoding. Existing conflict/modality error tests remain. Command projection is not a claim that a real runtime was installed/launched. Hosted checks explicitly run the facade and registry resolution tests. + +Formatting and whitespace pass locally. No local Rust execution is claimed. Independent source review and fresh hosted boundary/Clippy qualification are pending; public high-arity APIs remain separate. + +Independent source review accepted tree 29515000a6205a212139bcc53a6c65acd465e9d7. This covers only the two boxed concrete error boundaries and preserved payloads/branch order. Pure successful-command projection remains distinct from real runtime installation/readiness qualification; fresh hosted tests are required. From fb61592a2b1c14b9dedb13d5bdb1872ffcf192cf Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 05:13:21 -0700 Subject: [PATCH 08/25] refactor(inference): name PyTorch text generation inputs --- .github/workflows/quality-gates.yml | 7 + CHANGELOG.md | 4 + crates/inference/src/backend/mod.rs | 2 +- crates/inference/src/backend/pytorch.rs | 99 ++++------ crates/inference/src/backend/pytorch_tests.rs | 179 +++++++++++++++--- crates/inference/src/lib.rs | 2 +- ...6-10-03-pytorch-text-generation-request.md | 11 ++ 7 files changed, 217 insertions(+), 87 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-pytorch-text-generation-request.md diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml index 548da9f1e..1954ded40 100644 --- a/.github/workflows/quality-gates.yml +++ b/.github/workflows/quality-gates.yml @@ -218,6 +218,13 @@ jobs: cargo test -p inference --lib managed_binaries::tests cargo test -p inference --test model_contracts task_registry_resolution_ + - name: Run named PyTorch text request contract tests + run: | + cargo test -p inference --features backend-pytorch --lib pytorch_named_text_request -- --list > "$RUNNER_TEMP/pytorch-named-text-tests.list" + python3 -c 'import pathlib, sys; names = [line for line in pathlib.Path(sys.argv[1]).read_text().splitlines() if line.startswith("backend::pytorch::tests::pytorch_named_text_request") and line.endswith(": test")]; print("Named PyTorch text tests discovered:", len(names)); sys.exit(0 if names else 1)' "$RUNNER_TEMP/pytorch-named-text-tests.list" + cargo test -p inference --features backend-pytorch --lib pytorch_named_text_request + cargo test -p inference --features backend-pytorch --lib test_pytorch_generate_text_ + - name: Run node-engine tests run: cargo test -p node-engine --lib diff --git a/CHANGELOG.md b/CHANGELOG.md index d74a50be3..81822497d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,6 +14,10 @@ The format is based on Keep a Changelog. canonicalization records. ### Changed +- Rust source API: PyTorch `generate_with_top_k` and `generate_stream_with_top_k` + now accept `PyTorchTextGenerationRequest` with the same seven named inputs. + Legacy `generate`/`generate_stream` signatures and the worker wire contract + remain unchanged. - Rust source API: `resolve_managed_binary_command` and `resolve_task_registry_entry_from_evidence` now return boxed concrete errors. Error variants, diagnostic JSON and Display text remain unchanged; callers diff --git a/crates/inference/src/backend/mod.rs b/crates/inference/src/backend/mod.rs index 6da38cd6b..2b26c39e0 100644 --- a/crates/inference/src/backend/mod.rs +++ b/crates/inference/src/backend/mod.rs @@ -63,7 +63,7 @@ pub use llamacpp::LlamaCppBackend; pub use candle::CandleBackend; #[cfg(feature = "backend-pytorch")] -pub use pytorch::PyTorchBackend; +pub use pytorch::{PyTorchBackend, PyTorchTextGenerationRequest}; pub use compatibility::{ BackendCompatibilityIssue, BackendCompatibilityIssueKind, BackendCompatibilityOptions, diff --git a/crates/inference/src/backend/pytorch.rs b/crates/inference/src/backend/pytorch.rs index 52dae7fbe..55b455063 100644 --- a/crates/inference/src/backend/pytorch.rs +++ b/crates/inference/src/backend/pytorch.rs @@ -100,6 +100,19 @@ impl PyTorchDeviceProbeSnapshot { } } +/// Owned scalar inputs accepted by the PyTorch text generation entry points. +/// Worker-specific policy fields remain private to the worker contract. +#[derive(Debug, Clone, PartialEq)] +pub struct PyTorchTextGenerationRequest { + pub prompt: String, + pub system_prompt: Option, + pub max_tokens: i64, + pub temperature: f64, + pub top_p: f64, + pub top_k: Option, + pub masked_prompt_json: Option, +} + /// PyTorch backend using in-process PyO3 embedded Python. /// /// Loads models via HuggingFace transformers with `trust_remote_code=True`, @@ -2109,25 +2122,19 @@ impl PyTorchBackend { fn generate_text_envelope( request_id: impl Into, operation: PyTorchWorkerOperation, - prompt: String, - system_prompt: Option, - max_tokens: i64, - temperature: f64, - top_p: f64, - top_k: Option, - masked_prompt_json: Option, + request: PyTorchTextGenerationRequest, ) -> PyTorchWorkerEnvelope { PyTorchWorkerEnvelope::new( request_id, operation, Self::generate_text_request( - prompt, - system_prompt, - max_tokens, - temperature, - top_p, - top_k, - masked_prompt_json, + request.prompt, + request.system_prompt, + request.max_tokens, + request.temperature, + request.top_p, + request.top_k, + request.masked_prompt_json, ), ) } @@ -2492,39 +2499,27 @@ impl PyTorchBackend { top_p: f64, masked_prompt_json: Option, ) -> Result { - self.generate_with_top_k( + self.generate_with_top_k(PyTorchTextGenerationRequest { prompt, system_prompt, max_tokens, temperature, top_p, - None, + top_k: None, masked_prompt_json, - ) + }) .await } pub async fn generate_with_top_k( &self, - prompt: String, - system_prompt: Option, - max_tokens: i64, - temperature: f64, - top_p: f64, - top_k: Option, - masked_prompt_json: Option, + request: PyTorchTextGenerationRequest, ) -> Result { let request_id = format!("pytorch-generate-text-{}", Uuid::new_v4().simple()); let envelope = Self::generate_text_envelope( request_id.clone(), PyTorchWorkerOperation::GenerateText, - prompt, - system_prompt, - max_tokens, - temperature, - top_p, - top_k, - masked_prompt_json, + request, ); Self::validate_generate_text_envelope(&envelope)?; let envelope_json = serde_json::to_string(&envelope).map_err(|error| { @@ -2601,39 +2596,27 @@ impl PyTorchBackend { top_p: f64, masked_prompt_json: Option, ) -> Pin> + Send>> { - self.generate_stream_with_top_k( + self.generate_stream_with_top_k(PyTorchTextGenerationRequest { prompt, system_prompt, max_tokens, temperature, top_p, - None, + top_k: None, masked_prompt_json, - ) + }) } pub fn generate_stream_with_top_k( &self, - prompt: String, - system_prompt: Option, - max_tokens: i64, - temperature: f64, - top_p: f64, - top_k: Option, - masked_prompt_json: Option, + request: PyTorchTextGenerationRequest, ) -> Pin> + Send>> { let (tx, rx) = tokio::sync::mpsc::channel::>(32); let request_id = format!("pytorch-generate-text-stream-{}", Uuid::new_v4().simple()); let envelope = Self::generate_text_envelope( request_id.clone(), PyTorchWorkerOperation::GenerateTextStream, - prompt, - system_prompt, - max_tokens, - temperature, - top_p, - top_k, - masked_prompt_json, + request, ); if let Err(error) = Self::validate_generate_text_stream_envelope(&envelope) { @@ -2887,15 +2870,17 @@ impl InferenceBackend for PyTorchBackend { .and_then(|value| value.as_u64()) .and_then(|value| u32::try_from(value).ok()); - Ok(self.generate_stream_with_top_k( - prompt, - system_prompt, - max_tokens, - temperature, - top_p, - top_k, - None, - )) + Ok( + self.generate_stream_with_top_k(PyTorchTextGenerationRequest { + prompt, + system_prompt, + max_tokens, + temperature, + top_p, + top_k, + masked_prompt_json: None, + }), + ) } async fn embeddings( diff --git a/crates/inference/src/backend/pytorch_tests.rs b/crates/inference/src/backend/pytorch_tests.rs index 178f2f7ca..ca751e16a 100644 --- a/crates/inference/src/backend/pytorch_tests.rs +++ b/crates/inference/src/backend/pytorch_tests.rs @@ -2650,24 +2650,28 @@ fn test_pytorch_generate_text_envelopes_thread_top_k_for_generate_and_stream() { let generate_envelope = PyTorchBackend::generate_text_envelope( "req-generate-top-k", PyTorchWorkerOperation::GenerateText, - "Explain adapters.".to_string(), - Some("Be precise.".to_string()), - 48, - 0.3, - 0.9, - Some(33), - None, + PyTorchTextGenerationRequest { + prompt: "Explain adapters.".to_string(), + system_prompt: Some("Be precise.".to_string()), + max_tokens: 48, + temperature: 0.3, + top_p: 0.9, + top_k: Some(33), + masked_prompt_json: None, + }, ); let stream_envelope = PyTorchBackend::generate_text_envelope( "req-stream-top-k", PyTorchWorkerOperation::GenerateTextStream, - "Explain adapters.".to_string(), - Some("Be precise.".to_string()), - 48, - 0.3, - 0.9, - Some(33), - None, + PyTorchTextGenerationRequest { + prompt: "Explain adapters.".to_string(), + system_prompt: Some("Be precise.".to_string()), + max_tokens: 48, + temperature: 0.3, + top_p: 0.9, + top_k: Some(33), + masked_prompt_json: None, + }, ); PyTorchBackend::validate_generate_text_envelope(&generate_envelope) @@ -2689,13 +2693,15 @@ fn test_pytorch_generate_text_envelope_rejects_unscoped_transformers_kwargs() { let mut generate_envelope = PyTorchBackend::generate_text_envelope( "req-generate-raw-kwarg", PyTorchWorkerOperation::GenerateText, - "Explain adapters.".to_string(), - None, - 48, - 0.3, - 0.9, - None, - None, + PyTorchTextGenerationRequest { + prompt: "Explain adapters.".to_string(), + system_prompt: None, + max_tokens: 48, + temperature: 0.3, + top_p: 0.9, + top_k: None, + masked_prompt_json: None, + }, ); generate_envelope .payload @@ -2716,13 +2722,15 @@ fn test_pytorch_generate_text_stream_envelope_rejects_policy_transformers_kwargs let mut stream_envelope = PyTorchBackend::generate_text_envelope( "req-stream-policy-kwarg", PyTorchWorkerOperation::GenerateTextStream, - "Explain adapters.".to_string(), - None, - 48, - 0.3, - 0.9, - None, - None, + PyTorchTextGenerationRequest { + prompt: "Explain adapters.".to_string(), + system_prompt: None, + max_tokens: 48, + temperature: 0.3, + top_p: 0.9, + top_k: None, + masked_prompt_json: None, + }, ); stream_envelope .payload @@ -6398,3 +6406,118 @@ fn selected_text_adapter_preserves_text_parts_and_rejects_nontext_parts() { ]}]}); assert!(extract_prompt_from_messages(&request).is_err()); } + +fn named_text_request(prompt: &str) -> crate::PyTorchTextGenerationRequest { + crate::PyTorchTextGenerationRequest { + prompt: prompt.to_string(), + system_prompt: Some("Be concise.".to_string()), + max_tokens: 64, + temperature: 0.2, + top_p: 0.95, + top_k: Some(40), + masked_prompt_json: Some("{\"prompt\":\"masked\"}".to_string()), + } +} + +#[test] +fn pytorch_named_text_request_preserves_exact_worker_envelopes() { + for (operation, label) in [ + (PyTorchWorkerOperation::GenerateText, "generate_text"), + ( + PyTorchWorkerOperation::GenerateTextStream, + "generate_text_stream", + ), + ] { + let envelope = PyTorchBackend::generate_text_envelope( + "request-named", + operation, + named_text_request("Explain adapters."), + ); + PyTorchBackend::validate_generate_text_envelope_operation(&envelope, operation) + .expect("same worker validation"); + assert_eq!( + serde_json::to_value(&envelope).expect("worker envelope"), + serde_json::json!({ + "contract_version": 1, + "request_id": "request-named", + "operation": label, + "cancellation": { "drop_stream_cancels": false }, + "payload": { + "prompt": "Explain adapters.", + "system_prompt": "Be concise.", + "max_tokens": 64, + "temperature": 0.2, + "top_p": 0.95, + "masked_prompt_json": "{\"prompt\":\"masked\"}", + "transformers_kwargs": { "top_k": 40 }, + }, + }) + ); + } +} + +#[tokio::test] +async fn pytorch_named_text_request_preserves_legacy_validation_paths() { + use futures_util::StreamExt; + use std::time::Duration; + + let backend = PyTorchBackend::new(); + let mut request = named_text_request(" "); + request.top_k = None; + request.masked_prompt_json = None; + let legacy = backend + .generate( + " ".to_string(), + Some("Be concise.".to_string()), + 64, + 0.2, + 0.95, + None, + ) + .await + .expect_err("blank legacy prompt"); + let named = backend + .generate_with_top_k(request.clone()) + .await + .expect_err("blank named prompt"); + assert_eq!(legacy.to_string(), named.to_string()); + assert!( + matches!(named, BackendError::Config(ref message) if message == "PyTorch worker generate_text envelope requires a prompt") + ); + + let mut legacy_stream = backend.generate_stream( + " ".to_string(), + Some("Be concise.".to_string()), + 64, + 0.2, + 0.95, + None, + ); + let mut named_stream = backend.generate_stream_with_top_k(request); + let legacy_error = tokio::time::timeout(Duration::from_secs(2), legacy_stream.next()) + .await + .expect("bounded legacy validation") + .expect("legacy error item") + .expect_err("legacy validation failure"); + let named_error = tokio::time::timeout(Duration::from_secs(2), named_stream.next()) + .await + .expect("bounded named validation") + .expect("named error item") + .expect_err("named validation failure"); + assert_eq!(legacy_error.to_string(), named_error.to_string()); + assert!( + matches!(named_error, BackendError::Config(ref message) if message == "PyTorch worker generate_text envelope requires a prompt") + ); + assert!( + tokio::time::timeout(Duration::from_secs(2), legacy_stream.next()) + .await + .expect("legacy stream closes") + .is_none() + ); + assert!( + tokio::time::timeout(Duration::from_secs(2), named_stream.next()) + .await + .expect("named stream closes") + .is_none() + ); +} diff --git a/crates/inference/src/lib.rs b/crates/inference/src/lib.rs index a10466048..961841a4a 100644 --- a/crates/inference/src/lib.rs +++ b/crates/inference/src/lib.rs @@ -79,7 +79,7 @@ pub use backend::LlamaCppBackend; pub use backend::CandleBackend; #[cfg(feature = "backend-pytorch")] -pub use backend::PyTorchBackend; +pub use backend::{PyTorchBackend, PyTorchTextGenerationRequest}; pub use config::{DeviceConfig, EmbeddingMemoryMode}; pub use dependency_requirements::{ diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-pytorch-text-generation-request.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-pytorch-text-generation-request.md new file mode 100644 index 000000000..0ba10cc8d --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-pytorch-text-generation-request.md @@ -0,0 +1,11 @@ +# Named PyTorch text generation inputs + +The existing seven scalar inputs are grouped in public PyTorchTextGenerationRequest for the two top_k entry points and the private envelope helper. No existing request type represented exactly these inputs: the private worker DTO includes additional denoising/block controls and arbitrary kwargs, and general generation options carry broader policy. The new input type has no defaults or serialization contract and is re-exported beside PyTorchBackend under the existing feature gate. + +All three repository callers of the two public methods are migrated. Legacy generate/generate_stream signatures remain unchanged and wrap the same values. Strings move through the new container without added production clones; top_k still maps only to transformers_kwargs and absent values remain absent. The private worker DTO, request IDs, cancellation defaults, validation, stream/job behavior and Python worker are unchanged. The two changed public Rust call shapes are explicitly recorded in the changelog. + +Tests assert exact generated and streaming wire envelopes for all seven inputs, including the existing cancellation default, and compare legacy/named early validation errors and bounded stream closure. Existing top_k/unknown-policy envelope tests remain. Hosted commands explicitly enable backend-pytorch and require nonzero new-test discovery; no actual model inference is claimed by the validation-only tests. + +Formatting/whitespace pass locally. No local Rust execution is claimed. Independent source review and fresh hosted PyTorch contract/Clippy qualification remain pending. Public llama-server requests and other error owners are outside this slice. + +Independent source review accepted tree 6f40ff092f2b12780fb2152d0755ae3b7c531c23: exact owned inputs, unchanged legacy defaults/worker mapping, and bounded validation/stream tests. Fresh hosted execution remains required. The existing remote-code documentation/policy wording is outside this arity slice. From 9458fac38ad2d6bbbe4c17a565a9066d613a7fe9 Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 05:33:26 -0700 Subject: [PATCH 09/25] refactor(inference): reuse effective llama runtime settings --- .github/workflows/quality-gates.yml | 8 ++ CHANGELOG.md | 4 + crates/inference/src/backend/llamacpp.rs | 13 +-- crates/inference/src/server.rs | 100 +++++++++++------- crates/inference/src/server_tests.rs | 100 ++++++++---------- .../2026-10-03-llama-settings-input.md | 11 ++ 6 files changed, 131 insertions(+), 105 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-llama-settings-input.md diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml index 1954ded40..14752d008 100644 --- a/.github/workflows/quality-gates.yml +++ b/.github/workflows/quality-gates.yml @@ -225,6 +225,14 @@ jobs: cargo test -p inference --features backend-pytorch --lib pytorch_named_text_request cargo test -p inference --features backend-pytorch --lib test_pytorch_generate_text_ + - name: Run llama server startup and reuse contract tests + run: | + for test_filter in server::tests backend::llamacpp::tests; do + cargo test -p inference --features backend-llamacpp --lib "$test_filter" -- --list > "$RUNNER_TEMP/llama-settings-tests.list" + python3 -c 'import pathlib, sys; prefix = sys.argv[2] + "::"; names = [line for line in pathlib.Path(sys.argv[1]).read_text().splitlines() if line.startswith(prefix) and line.endswith(": test")]; print(prefix, "tests discovered:", len(names)); sys.exit(0 if names else 1)' "$RUNNER_TEMP/llama-settings-tests.list" "$test_filter" + cargo test -p inference --features backend-llamacpp --lib "$test_filter" + done + - name: Run node-engine tests run: cargo test -p node-engine --lib diff --git a/CHANGELOG.md b/CHANGELOG.md index 81822497d..f7039893a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,6 +14,10 @@ The format is based on Keep a Changelog. canonicalization records. ### Changed +- Rust source API: `LlamaServer::start_sidecar_inference` and + `matches_inference_runtime` now borrow the existing `LlamaCppRuntimeSettings` + for effective device/context/thread/batch values. Model/mmproj paths, process + spawner and port override retain their existing roles and ownership. - Rust source API: PyTorch `generate_with_top_k` and `generate_stream_with_top_k` now accept `PyTorchTextGenerationRequest` with the same seven named inputs. Legacy `generate`/`generate_stream` signatures and the worker wire contract diff --git a/crates/inference/src/backend/llamacpp.rs b/crates/inference/src/backend/llamacpp.rs index 783487742..99c19ee42 100644 --- a/crates/inference/src/backend/llamacpp.rs +++ b/crates/inference/src/backend/llamacpp.rs @@ -212,7 +212,6 @@ impl InferenceBackend for LlamaCppBackend { let runtime_settings = LlamaCppRuntimeSettings::try_from_backend_config(config)?; let device_config = runtime_settings.device_config(); - let context_size = runtime_settings.context_size; if config.embedding_mode { // Start in embedding mode @@ -289,11 +288,7 @@ impl InferenceBackend for LlamaCppBackend { if self.server.matches_inference_runtime( &model_path.to_string_lossy(), mmproj_path.as_deref(), - &device_config, - context_size, - runtime_settings.cpu_threads, - runtime_settings.batch_size, - runtime_settings.ubatch_size, + &runtime_settings, config.port_override, ) { return Ok(BackendStartOutcome { @@ -307,11 +302,7 @@ impl InferenceBackend for LlamaCppBackend { spawner, &model_path.to_string_lossy(), mmproj_path.as_deref(), - &device_config, - context_size, - runtime_settings.cpu_threads, - runtime_settings.batch_size, - runtime_settings.ubatch_size, + &runtime_settings, config.port_override, ) .await diff --git a/crates/inference/src/server.rs b/crates/inference/src/server.rs index 2de053845..50cb2cfbb 100644 --- a/crates/inference/src/server.rs +++ b/crates/inference/src/server.rs @@ -12,6 +12,7 @@ use std::sync::Arc; use sysinfo::{Pid, ProcessesToUpdate, Signal, System}; use tokio::sync::RwLock; +use crate::backend::LlamaCppRuntimeSettings; use crate::config::DeviceConfig; use crate::constants::{hosts, ports, timeouts}; use crate::llamacpp_sidecar_events::{ @@ -54,17 +55,32 @@ fn parse_sidecar_pid(raw: &str) -> Option { .map(|record| record.pid) } -fn active_runtime_descriptor( +struct RuntimeDescriptorInput<'a> { mode: LlamaCppRuntimeMode, port: u16, - model_path: &str, - mmproj_path: Option<&str>, - device: &DeviceConfig, + model_path: &'a str, + mmproj_path: Option<&'a str>, + device: &'a DeviceConfig, context_size: Option, cpu_threads: Option, batch_size: Option, ubatch_size: Option, +} + +fn active_runtime_descriptor( + input: RuntimeDescriptorInput<'_>, ) -> Option { + let RuntimeDescriptorInput { + mode, + port, + model_path, + mmproj_path, + device, + context_size, + cpu_threads, + batch_size, + ubatch_size, + } = input; let selected_device = selected_contract_device(device)?; let (selected_device_class, selected_device_id) = selected_device .map(|(device_class, device_id)| (Some(device_class), Some(device_id))) @@ -250,13 +266,14 @@ impl LlamaServer { spawner: Arc, model_path: &str, mmproj_path: Option<&str>, - device: &DeviceConfig, - context_size: u32, - cpu_threads: Option, - batch_size: Option, - ubatch_size: Option, + settings: &LlamaCppRuntimeSettings, port_override: Option, ) -> Result<(), LlamaCppSidecarStartupError> { + let device = &settings.device_config(); + let context_size = settings.context_size; + let cpu_threads = settings.cpu_threads; + let batch_size = settings.batch_size; + let ubatch_size = settings.ubatch_size; // Stop any existing connection self.stop(); @@ -636,13 +653,14 @@ impl LlamaServer { &self, model_path: &str, mmproj_path: Option<&str>, - device: &DeviceConfig, - context_size: u32, - cpu_threads: Option, - batch_size: Option, - ubatch_size: Option, + settings: &LlamaCppRuntimeSettings, port_override: Option, ) -> bool { + let device = &settings.device_config(); + let context_size = settings.context_size; + let cpu_threads = settings.cpu_threads; + let batch_size = settings.batch_size; + let ubatch_size = settings.ubatch_size; let expected_port = port_override.unwrap_or(ports::SERVER); let Some(active) = self.active_runtime_descriptor() else { return false; @@ -719,47 +737,47 @@ impl LlamaServer { cpu_threads, batch_size, ubatch_size, - } => active_runtime_descriptor( - LlamaCppRuntimeMode::Inference, - *port, + } => active_runtime_descriptor(RuntimeDescriptorInput { + mode: LlamaCppRuntimeMode::Inference, + port: *port, model_path, - mmproj_path.as_deref(), + mmproj_path: mmproj_path.as_deref(), device, - Some(*context_size), - *cpu_threads, - *batch_size, - *ubatch_size, - ), + context_size: Some(*context_size), + cpu_threads: *cpu_threads, + batch_size: *batch_size, + ubatch_size: *ubatch_size, + }), ServerMode::SidecarEmbedding { port, model_path, device, - } => active_runtime_descriptor( - LlamaCppRuntimeMode::Embedding, - *port, + } => active_runtime_descriptor(RuntimeDescriptorInput { + mode: LlamaCppRuntimeMode::Embedding, + port: *port, model_path, - None, + mmproj_path: None, device, - None, - None, - None, - None, - ), + context_size: None, + cpu_threads: None, + batch_size: None, + ubatch_size: None, + }), ServerMode::SidecarReranking { port, model_path, device, - } => active_runtime_descriptor( - LlamaCppRuntimeMode::Reranking, - *port, + } => active_runtime_descriptor(RuntimeDescriptorInput { + mode: LlamaCppRuntimeMode::Reranking, + port: *port, model_path, - None, + mmproj_path: None, device, - None, - None, - None, - None, - ), + context_size: None, + cpu_threads: None, + batch_size: None, + ubatch_size: None, + }), ServerMode::None | ServerMode::External { .. } => None, } } diff --git a/crates/inference/src/server_tests.rs b/crates/inference/src/server_tests.rs index 08e13b685..a4c03d619 100644 --- a/crates/inference/src/server_tests.rs +++ b/crates/inference/src/server_tests.rs @@ -1,4 +1,5 @@ use super::{parse_sidecar_pid, LlamaServer, ServerMode}; +use crate::backend::LlamaCppRuntimeSettings; use crate::config::DeviceConfig; use crate::device::DeviceBackend; use crate::llamacpp_sidecar_events::LlamaCppSidecarStartupError; @@ -89,21 +90,13 @@ fn inference_runtime_matcher_requires_matching_port() { assert!(server.matches_inference_runtime( "/models/main.gguf", Some("/models/vision.mmproj"), - &device, - 4096, - Some(8), - Some(512), - Some(128), + &inference_settings(&device, 4096, Some(8), Some(512), Some(128)), Some(11434), )); assert!(!server.matches_inference_runtime( "/models/other.gguf", Some("/models/vision.mmproj"), - &device, - 4096, - Some(8), - Some(512), - Some(128), + &inference_settings(&device, 4096, Some(8), Some(512), Some(128)), Some(11434), )); assert!(!server.matches_embedding_runtime("/models/main.gguf", &device, Some(11434),)); @@ -111,31 +104,19 @@ fn inference_runtime_matcher_requires_matching_port() { assert!(!server.matches_inference_runtime( "/models/main.gguf", Some("/models/vision.mmproj"), - &device, - 4096, - Some(8), - Some(512), - Some(128), + &inference_settings(&device, 4096, Some(8), Some(512), Some(128)), Some(18080), )); assert!(!server.matches_inference_runtime( "/models/main.gguf", Some("/models/vision.mmproj"), - &device, - 8192, - Some(8), - Some(512), - Some(128), + &inference_settings(&device, 8192, Some(8), Some(512), Some(128)), Some(11434), )); assert!(!server.matches_inference_runtime( "/models/main.gguf", Some("/models/vision.mmproj"), - &device, - 4096, - Some(16), - Some(512), - Some(128), + &inference_settings(&device, 4096, Some(16), Some(512), Some(128)), Some(11434), )); } @@ -356,14 +337,16 @@ async fn start_sidecar_inference_cleans_process_and_pid_file_on_start_error() { }), "/models/main.gguf", None, - &DeviceConfig { - device: DeviceBackend::Auto, - gpu_layers: -1, - }, - 4096, - None, - None, - None, + &inference_settings( + &DeviceConfig { + device: DeviceBackend::Auto, + gpu_layers: -1, + }, + 4096, + None, + None, + None, + ), Some(18080), ) .await; @@ -394,14 +377,16 @@ async fn start_sidecar_inference_applies_runtime_settings_to_llama_server_args() }), "/models/main.gguf", Some("/models/mmproj.gguf"), - &DeviceConfig { - device: DeviceBackend::Vulkan(0), - gpu_layers: 12, - }, - 16384, - Some(8), - Some(512), - Some(128), + &inference_settings( + &DeviceConfig { + device: DeviceBackend::Vulkan(0), + gpu_layers: 12, + }, + 16384, + Some(8), + Some(512), + Some(128), + ), Some(18080), ) .await; @@ -448,22 +433,14 @@ fn runtime_matchers_preserve_path_component_comparison() { assert!(server.matches_inference_runtime( "models/./main.gguf", None, - &device, - 4096, - None, - None, - None, - Some(11434) + &inference_settings(&device, 4096, None, None, None), + Some(11434), )); assert!(!server.matches_inference_runtime( "models/other.gguf", None, - &device, - 4096, - None, - None, - None, - Some(11434) + &inference_settings(&device, 4096, None, None, None), + Some(11434), )); server.set_test_runtime_state( ServerMode::SidecarEmbedding { @@ -486,3 +463,20 @@ fn runtime_matchers_preserve_path_component_comparison() { assert!(server.matches_reranking_runtime("models/./main.gguf", &device, Some(11434))); assert!(!server.matches_reranking_runtime("models/other.gguf", &device, Some(11434))); } + +fn inference_settings( + device: &DeviceConfig, + context_size: u32, + cpu_threads: Option, + batch_size: Option, + ubatch_size: Option, +) -> LlamaCppRuntimeSettings { + LlamaCppRuntimeSettings { + device: device.device.clone(), + gpu_layers: device.gpu_layers, + context_size, + cpu_threads, + batch_size, + ubatch_size, + } +} diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-llama-settings-input.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-llama-settings-input.md new file mode 100644 index 000000000..9c4a17f1b --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-llama-settings-input.md @@ -0,0 +1,11 @@ +# Llama server effective settings input + +Reuse existing backend-owned LlamaCppRuntimeSettings in the two public inference startup/reuse methods instead of repeating device/context/thread/batch parameters. The backend passes the same normalized settings object it already owns; the server projects the existing DeviceConfig and scalar values without new normalization or validation. Model/mmproj borrows, process spawner, port override/default, startup cleanup, and reuse semantics remain unchanged. Embedding/reranking public APIs are untouched. + +A private RuntimeDescriptorInput groups the existing nine snapshot parameters at all three callers. The descriptor construction block, startup behavior block and reuse comparison block are byte-identical after input destructuring/projection. Known call sites are one backend startup/reuse path and the server tests; Tauri's same-named command goes through the gateway. The two public Rust signature changes are recorded explicitly in the changelog. + +Existing tests retain exact launch settings/argument boundaries, startup process/PID cleanup, readiness/descriptor identities, port/context/device reuse distinctions and Path-component equality. Hosted commands enable backend-llamacpp, require nonzero discovery, and run both server and backend test modules. The tests use existing mock process infrastructure, not a claim of real model loading. + +Formatting, whitespace and unchanged-body comparisons pass locally. No local Rust execution is claimed. Independent source review and fresh hosted startup/reuse/Clippy qualification remain pending. This adds no runtime selection policy or scheduler algorithm. + +Independent source review accepted exact tree 47ac0a3557868b5d0847c84e151798c061ca2d8f with no blockers. It verified all callers and unchanged behavior blocks, plus 12 server and 11 backend test attributes in source and feature-enabled nonzero discovery. Those source counts are not execution receipts; fresh hosted tests remain required. From c277a0b26d94e6ad1b1d25cb7ff6aa54db6f79b7 Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 05:39:11 -0700 Subject: [PATCH 10/25] docs(inference): clarify default-deny remote code policy --- crates/inference/src/backend/pytorch.rs | 5 +++-- .../reports/2026-10-03-pytorch-trust-policy-docs.md | 9 +++++++++ 2 files changed, 12 insertions(+), 2 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-pytorch-trust-policy-docs.md diff --git a/crates/inference/src/backend/pytorch.rs b/crates/inference/src/backend/pytorch.rs index 55b455063..48271cdb3 100644 --- a/crates/inference/src/backend/pytorch.rs +++ b/crates/inference/src/backend/pytorch.rs @@ -115,8 +115,9 @@ pub struct PyTorchTextGenerationRequest { /// PyTorch backend using in-process PyO3 embedded Python. /// -/// Loads models via HuggingFace transformers with `trust_remote_code=True`, -/// supporting standard models, dLLM architectures, and Sherry quantised models. +/// Loads models via HuggingFace transformers using explicit model-load security +/// policy; custom remote code is denied by default. Supports standard models, +/// dLLM architectures, and Sherry quantised models. pub struct PyTorchBackend { /// Whether the backend has been initialised and is ready ready: bool, diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-pytorch-trust-policy-docs.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-pytorch-trust-policy-docs.md new file mode 100644 index 000000000..0a9cbff16 --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-pytorch-trust-policy-docs.md @@ -0,0 +1,9 @@ +# PyTorch trust-policy documentation correction + +The PyTorchBackend class comment incorrectly described unconditional trust_remote_code=True. The actual loader chain is policy-driven and defaults closed: ModelRemoteCodePolicy defaults to Deny; PyTorchTransformersTrustPolicy derives allow_remote_code only from explicit Allow; worker_contract.py defaults the missing flag to false; worker.py defaults trust_remote_code=False and rejects required custom code when it remains closed. + +Correct only the class documentation to describe that existing behavior. No runtime, policy default, validation, accepted-source, worker, or credential behavior changes. This is a source-grounded documentation correction, not a comprehensive security audit or authorization to load custom code. + +Whitespace/formatting checks pass. Runtime tests were not rerun for this comment-only change. Publication receipt pending. + +Source review accepted the exact documentation-only checkpoint 52e511f44a5de9113b7ec32c5f2df4f4935cbc34. The review confirms documentation accuracy only; no additional runtime execution is claimed. Publication follows the separately reviewed llama milestone; no runtime implementation changed in this slice. From 92ca20540422bbc74fb888c810893d3ce2121dbd Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 05:56:06 -0700 Subject: [PATCH 11/25] refactor(node-engine): group validation lifecycle attribution --- .github/workflows/quality-gates.yml | 6 +++ crates/node-engine/src/core_executor.rs | 45 +++++++++++++------ .../2026-10-03-node-validation-context.md | 7 +++ 3 files changed, 44 insertions(+), 14 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-node-validation-context.md diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml index 14752d008..aa09628bd 100644 --- a/.github/workflows/quality-gates.yml +++ b/.github/workflows/quality-gates.yml @@ -233,6 +233,12 @@ jobs: cargo test -p inference --features backend-llamacpp --lib "$test_filter" done + - name: Run contract-only node lifecycle tests + run: | + cargo test -p node-engine --features inference-nodes --lib rejects_contract_only_with_lifecycle -- --list > "$RUNNER_TEMP/node-validation-lifecycle-tests.list" + python3 -c 'import pathlib, sys; names = [line for line in pathlib.Path(sys.argv[1]).read_text().splitlines() if line.startswith("core_executor::tests::inference_tests::") and "rejects_contract_only_with_lifecycle" in line and line.endswith(": test")]; print("Contract-only lifecycle tests discovered:", len(names)); sys.exit(0 if names else 1)' "$RUNNER_TEMP/node-validation-lifecycle-tests.list" + cargo test -p node-engine --features inference-nodes --lib rejects_contract_only_with_lifecycle + - name: Run node-engine tests run: cargo test -p node-engine --lib diff --git a/crates/node-engine/src/core_executor.rs b/crates/node-engine/src/core_executor.rs index ef2192afb..2a79bac79 100644 --- a/crates/node-engine/src/core_executor.rs +++ b/crates/node-engine/src/core_executor.rs @@ -434,14 +434,16 @@ fn reject_contract_only_inference_task( let artifact_refs = contract_only_task_artifact_refs(inputs); record_task_validation_failure_lifecycle( extensions, - task_id, - execution_id, - entry.canonical_label(), - backend_key, - inference_model_id_from_inputs(inputs), - message.clone(), - option_diagnostics, - artifact_refs, + TaskValidationFailureContext { + task_id, + execution_id, + task_label: entry.canonical_label(), + backend_key, + model_id: inference_model_id_from_inputs(inputs), + detail: message.clone(), + option_diagnostics, + artifact_refs, + }, ); Err(NodeEngineError::ExecutionFailed(message)) @@ -570,17 +572,32 @@ fn task_option_present(inputs: &HashMap, aliases: &[& } #[cfg(feature = "inference-nodes")] -fn record_task_validation_failure_lifecycle( - extensions: &ExecutorExtensions, - task_id: &str, - execution_id: &str, - task_label: &str, - backend_key: Option<&str>, +struct TaskValidationFailureContext<'a> { + task_id: &'a str, + execution_id: &'a str, + task_label: &'a str, + backend_key: Option<&'a str>, model_id: Option, detail: String, option_diagnostics: Vec, artifact_refs: Vec, +} + +#[cfg(feature = "inference-nodes")] +fn record_task_validation_failure_lifecycle( + extensions: &ExecutorExtensions, + context: TaskValidationFailureContext<'_>, ) { + let TaskValidationFailureContext { + task_id, + execution_id, + task_label, + backend_key, + model_id, + detail, + option_diagnostics, + artifact_refs, + } = context; let Some(sink) = extensions .get::>( extension_keys::INFERENCE_LIFECYCLE_SINK, diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-node-validation-context.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-node-validation-context.md new file mode 100644 index 000000000..a14e36be1 --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-node-validation-context.md @@ -0,0 +1,7 @@ +# Private node validation lifecycle context + +Fresh PR #30 aggregate Clippy cleared inference and stopped at node-engine's private record_task_validation_failure_lifecycle helper. Group its existing eight values into TaskValidationFailureContext, keep ExecutorExtensions separate, and migrate the single caller. Both the new private type and helper retain the inference-nodes feature gate. The sink lookup and complete event loop are byte-identical after destructuring. + +The two existing contract-only depth/video tests already assert complete request/task/backend/runtime/model attribution, Started/Failed/Cleanup ordering, and failure-only option diagnostics and artifact references. Hosted commands now explicitly enable inference-nodes, require nonzero exact-module discovery, and execute these tests. Public API, task support policy, generated request IDs, diagnostics and sink failure handling are unchanged. + +Formatting, whitespace and unchanged event-body comparison pass locally. No local Rust execution is claimed. Root source review accepted frozen tree f20d0e683c0057dfa7337abe8b0e56f6e2a6effa after inspecting the exact three-file diff, unchanged eight-value grouping, feature gates, single caller and nonzero test discovery. This is source review, not Rust execution or independent-agent review. Fresh hosted tests/Clippy qualification and the full combined external review remain pending; no lint suppression is added. From a6fa0755f2b2f25b25d556e00bcb8f2da60ae234 Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 06:07:55 -0700 Subject: [PATCH 12/25] test(node-engine): disambiguate mock backend result type --- crates/node-engine/src/core_executor/kv_cache_test_support.rs | 2 +- .../reports/2026-10-03-node-validation-context.md | 4 ++++ 2 files changed, 5 insertions(+), 1 deletion(-) diff --git a/crates/node-engine/src/core_executor/kv_cache_test_support.rs b/crates/node-engine/src/core_executor/kv_cache_test_support.rs index eb2f56b59..8b90349fd 100644 --- a/crates/node-engine/src/core_executor/kv_cache_test_support.rs +++ b/crates/node-engine/src/core_executor/kv_cache_test_support.rs @@ -81,7 +81,7 @@ impl InferenceBackend for MockKvBackend { Ok(BackendStartOutcome::default()) } - async fn stop(&mut self) -> Result<(), BackendError> { + async fn stop(&mut self) -> std::result::Result<(), BackendError> { Ok(()) } diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-node-validation-context.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-node-validation-context.md index a14e36be1..995569d94 100644 --- a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-node-validation-context.md +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-node-validation-context.md @@ -5,3 +5,7 @@ Fresh PR #30 aggregate Clippy cleared inference and stopped at node-engine's pri The two existing contract-only depth/video tests already assert complete request/task/backend/runtime/model attribution, Started/Failed/Cleanup ordering, and failure-only option diagnostics and artifact references. Hosted commands now explicitly enable inference-nodes, require nonzero exact-module discovery, and execute these tests. Public API, task support policy, generated request IDs, diagnostics and sink failure handling are unchanged. Formatting, whitespace and unchanged event-body comparison pass locally. No local Rust execution is claimed. Root source review accepted frozen tree f20d0e683c0057dfa7337abe8b0e56f6e2a6effa after inspecting the exact three-file diff, unchanged eight-value grouping, feature gates, single caller and nonzero test discovery. This is source review, not Rust execution or independent-agent review. Fresh hosted tests/Clippy qualification and the full combined external review remain pending; no lint suppression is added. + +## Feature-enabled fixture compilation correction + +The initial exact-head focused job 111207010012 at 92ca205 failed before lifecycle test discovery: MockKvBackend::stop resolved Result through super::* to node-engine's one-parameter NodeEngineError alias (E0107/E0053). This fixture file is unchanged from main 4938e405. Qualify std::result::Result explicitly, matching InferenceBackend::stop and neighboring mock methods, without changing the successful mock response. The two lifecycle tests have not yet executed; the feature/nonzero gate remains in place. Root source review accepted correction tree 39a1d45aaf80fe05264f5f26357e3cb2d00a9875. Corrected hosted execution remains pending. From 3c4dbbd331a83ffd5f249392eda5dfd47ee3142c Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 06:12:33 -0700 Subject: [PATCH 13/25] refactor(runtime-host): implement standard borrowed accessors --- .github/workflows/quality-gates.yml | 4 ++ CHANGELOG.md | 4 ++ .../src/reservation_lifecycle.rs | 14 +++-- .../src/runtime_host_execution.rs | 28 ++++++---- .../src/runtime_session_load.rs | 7 ++- .../tests/as_ref_compatibility.rs | 56 +++++++++++++++++++ .../reports/2026-10-03-runtime-host-asref.md | 7 +++ 7 files changed, 99 insertions(+), 21 deletions(-) create mode 100644 crates/pantograph-runtime-host-contracts/tests/as_ref_compatibility.rs create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-runtime-host-asref.md diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml index aa09628bd..9b767666e 100644 --- a/.github/workflows/quality-gates.yml +++ b/.github/workflows/quality-gates.yml @@ -190,6 +190,10 @@ jobs: - name: Check scheduler standard-trait accessor lint run: cargo clippy -p pantograph-scheduler --all-targets -- -D clippy::should_implement_trait + - name: Run runtime-host contract and accessor compatibility tests + run: cargo test -p pantograph-runtime-host-contracts + - name: Check runtime-host standard-trait accessor lint + run: cargo clippy -p pantograph-runtime-host-contracts --all-targets -- -D clippy::should_implement_trait - name: Run media conversion contract tests run: cargo test -p pantograph-media-conversion diff --git a/CHANGELOG.md b/CHANGELOG.md index f7039893a..67e487d99 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -47,3 +47,7 @@ The format is based on Keep a Changelog. ### Security - Canonical path validation enforced at file-boundary entry points to block traversal and symlink escape paths. + +### Runtime-host borrowed accessor compatibility + +The seven validated runtime-host contract wrappers now implement standard `AsRef` in place of inherent `as_ref` methods. Borrowed values and lifetimes, wrapper types, validation and `into_inner` are unchanged. Ordinary method syntax, qualified calls and function pointers continue to resolve through the standard prelude. Rust callers using `no_implicit_prelude` must explicitly import `std::convert::AsRef`. The workspace crate remains unpublished (`publish = false`). diff --git a/crates/pantograph-runtime-host-contracts/src/reservation_lifecycle.rs b/crates/pantograph-runtime-host-contracts/src/reservation_lifecycle.rs index 68876e834..c22409ae2 100644 --- a/crates/pantograph-runtime-host-contracts/src/reservation_lifecycle.rs +++ b/crates/pantograph-runtime-host-contracts/src/reservation_lifecycle.rs @@ -133,12 +133,13 @@ impl ReservationLifecycleEvent { #[must_use] pub struct ValidatedReservationLifecycleEvent(ReservationLifecycleEvent); -impl ValidatedReservationLifecycleEvent { - #[must_use] - pub fn as_ref(&self) -> &ReservationLifecycleEvent { +impl AsRef for ValidatedReservationLifecycleEvent { + fn as_ref(&self) -> &ReservationLifecycleEvent { &self.0 } +} +impl ValidatedReservationLifecycleEvent { #[must_use] pub fn into_inner(self) -> ReservationLifecycleEvent { self.0 @@ -195,12 +196,13 @@ impl ReservationLifecycleApplication { #[must_use] pub struct ValidatedReservationLifecycleApplication(ReservationLifecycleApplication); -impl ValidatedReservationLifecycleApplication { - #[must_use] - pub fn as_ref(&self) -> &ReservationLifecycleApplication { +impl AsRef for ValidatedReservationLifecycleApplication { + fn as_ref(&self) -> &ReservationLifecycleApplication { &self.0 } +} +impl ValidatedReservationLifecycleApplication { #[must_use] pub fn into_inner(self) -> ReservationLifecycleApplication { self.0 diff --git a/crates/pantograph-runtime-host-contracts/src/runtime_host_execution.rs b/crates/pantograph-runtime-host-contracts/src/runtime_host_execution.rs index ae267ba72..74506beff 100644 --- a/crates/pantograph-runtime-host-contracts/src/runtime_host_execution.rs +++ b/crates/pantograph-runtime-host-contracts/src/runtime_host_execution.rs @@ -167,12 +167,13 @@ impl RuntimeHostExecutionInputValue { #[must_use] pub struct ValidatedRuntimeHostExecutionRequest(RuntimeHostExecutionRequest); -impl ValidatedRuntimeHostExecutionRequest { - #[must_use] - pub fn as_ref(&self) -> &RuntimeHostExecutionRequest { +impl AsRef for ValidatedRuntimeHostExecutionRequest { + fn as_ref(&self) -> &RuntimeHostExecutionRequest { &self.0 } +} +impl ValidatedRuntimeHostExecutionRequest { #[must_use] pub fn into_inner(self) -> RuntimeHostExecutionRequest { self.0 @@ -388,12 +389,13 @@ impl RuntimeHostExecutionResponse { #[must_use] pub struct ValidatedRuntimeHostExecutionResponse(RuntimeHostExecutionResponse); -impl ValidatedRuntimeHostExecutionResponse { - #[must_use] - pub fn as_ref(&self) -> &RuntimeHostExecutionResponse { +impl AsRef for ValidatedRuntimeHostExecutionResponse { + fn as_ref(&self) -> &RuntimeHostExecutionResponse { &self.0 } +} +impl ValidatedRuntimeHostExecutionResponse { #[must_use] pub fn into_inner(self) -> RuntimeHostExecutionResponse { self.0 @@ -724,12 +726,13 @@ pub enum RuntimeHostBatchMemberReservationDisposition { #[must_use] pub struct ValidatedRuntimeHostBatchExecutionRequest(RuntimeHostBatchExecutionRequest); -impl ValidatedRuntimeHostBatchExecutionRequest { - #[must_use] - pub fn as_ref(&self) -> &RuntimeHostBatchExecutionRequest { +impl AsRef for ValidatedRuntimeHostBatchExecutionRequest { + fn as_ref(&self) -> &RuntimeHostBatchExecutionRequest { &self.0 } +} +impl ValidatedRuntimeHostBatchExecutionRequest { #[must_use] pub fn into_inner(self) -> RuntimeHostBatchExecutionRequest { self.0 @@ -749,12 +752,13 @@ impl TryFrom for ValidatedRuntimeHostBatchExec #[must_use] pub struct ValidatedRuntimeHostBatchExecutionResponse(RuntimeHostBatchExecutionResponse); -impl ValidatedRuntimeHostBatchExecutionResponse { - #[must_use] - pub fn as_ref(&self) -> &RuntimeHostBatchExecutionResponse { +impl AsRef for ValidatedRuntimeHostBatchExecutionResponse { + fn as_ref(&self) -> &RuntimeHostBatchExecutionResponse { &self.0 } +} +impl ValidatedRuntimeHostBatchExecutionResponse { #[must_use] pub fn into_inner(self) -> RuntimeHostBatchExecutionResponse { self.0 diff --git a/crates/pantograph-runtime-host-contracts/src/runtime_session_load.rs b/crates/pantograph-runtime-host-contracts/src/runtime_session_load.rs index 30a5a48c1..3005ab079 100644 --- a/crates/pantograph-runtime-host-contracts/src/runtime_session_load.rs +++ b/crates/pantograph-runtime-host-contracts/src/runtime_session_load.rs @@ -89,12 +89,13 @@ pub enum WorkflowSessionRuntimeLoadProofDiagnosticPhase { #[must_use] pub struct ValidatedWorkflowSessionRuntimeLoadProof(WorkflowSessionRuntimeLoadProof); -impl ValidatedWorkflowSessionRuntimeLoadProof { - #[must_use] - pub fn as_ref(&self) -> &WorkflowSessionRuntimeLoadProof { +impl AsRef for ValidatedWorkflowSessionRuntimeLoadProof { + fn as_ref(&self) -> &WorkflowSessionRuntimeLoadProof { &self.0 } +} +impl ValidatedWorkflowSessionRuntimeLoadProof { #[must_use] pub fn into_inner(self) -> WorkflowSessionRuntimeLoadProof { self.0 diff --git a/crates/pantograph-runtime-host-contracts/tests/as_ref_compatibility.rs b/crates/pantograph-runtime-host-contracts/tests/as_ref_compatibility.rs new file mode 100644 index 000000000..4cd23823c --- /dev/null +++ b/crates/pantograph-runtime-host-contracts/tests/as_ref_compatibility.rs @@ -0,0 +1,56 @@ +//! Compile coverage of public accessor call forms after migration to AsRef. +use pantograph_runtime_host_contracts::*; + +macro_rules! accessor_call_forms { + ($name:ident, $wrapper:ty, $raw:ty) => { + #[test] + fn $name() { + let _: fn(&$wrapper) -> &$raw = |value| value.as_ref(); + let _: fn(&$wrapper) -> &$raw = |value| <$wrapper>::as_ref(value); + let _: fn(&$wrapper) -> &$raw = <$wrapper>::as_ref; + let _: fn(&$wrapper) -> &$raw = |value| <$wrapper as AsRef<$raw>>::as_ref(value); + } + }; +} + +accessor_call_forms!( + validated_reservation_lifecycle_event, + ValidatedReservationLifecycleEvent, + ReservationLifecycleEvent +); + +accessor_call_forms!( + validated_reservation_lifecycle_application, + ValidatedReservationLifecycleApplication, + ReservationLifecycleApplication +); + +accessor_call_forms!( + validated_runtime_host_execution_request, + ValidatedRuntimeHostExecutionRequest, + RuntimeHostExecutionRequest +); + +accessor_call_forms!( + validated_runtime_host_execution_response, + ValidatedRuntimeHostExecutionResponse, + RuntimeHostExecutionResponse +); + +accessor_call_forms!( + validated_runtime_host_batch_execution_request, + ValidatedRuntimeHostBatchExecutionRequest, + RuntimeHostBatchExecutionRequest +); + +accessor_call_forms!( + validated_runtime_host_batch_execution_response, + ValidatedRuntimeHostBatchExecutionResponse, + RuntimeHostBatchExecutionResponse +); + +accessor_call_forms!( + validated_workflow_session_runtime_load_proof, + ValidatedWorkflowSessionRuntimeLoadProof, + WorkflowSessionRuntimeLoadProof +); diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-runtime-host-asref.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-runtime-host-asref.md new file mode 100644 index 000000000..638f26128 --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-runtime-host-asref.md @@ -0,0 +1,7 @@ +# Runtime-host standard borrowed accessors + +Fresh PR #32 Clippy job 111207010086 reports seven should_implement_trait errors in runtime-host-contracts. Move exactly those borrowed accessors into AsRef implementations, preserving the seven exported wrapper/raw pairs, validation, constructors, into_inner and raw reference lifetimes. Retain wrapper must_use; the redundant inherent accessor annotation disappears with the method migration. + +The source inventory covers runtime-host-contracts, embedded-runtime and workflow-service callers. No existing qualified accessor/function-pointer or no_implicit_prelude callers were found. Seven compile tests cover method syntax, qualified syntax, a typed function pointer and fully qualified trait syntax. Existing contract behavior tests remain, and hosted CI adds the complete crate suite plus the targeted trait lint. PR #19 already established with a Rust 1.92 probe that adding the trait while retaining the inherent method does not satisfy this lint. + +The public compatibility note records the explicit trait import needed for no_implicit_prelude consumers. This publish=false crate has no version bump. Formatting and staged whitespace are checked locally; root source review accepted frozen tree a8fc84c89cf8c97198ac2e2e4fb0ee437782a228 after inspecting all seven files; fresh hosted execution remains pending. No local workspace Rust qualification is claimed. From 395596fb0daf91b07d0075d0df51f22061597be1 Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 06:20:35 -0700 Subject: [PATCH 14/25] refactor(diagnostics): reuse projection state input --- .github/workflows/quality-gates.yml | 2 + .../src/sqlite/event_sqlite.rs | 110 +++++++++++------- .../src/tests.rs | 9 ++ ...6-10-03-ledger-private-projection-input.md | 7 ++ 4 files changed, 87 insertions(+), 41 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-ledger-private-projection-input.md diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml index 9b767666e..9963b1131 100644 --- a/.github/workflows/quality-gates.yml +++ b/.github/workflows/quality-gates.yml @@ -194,6 +194,8 @@ jobs: run: cargo test -p pantograph-runtime-host-contracts - name: Check runtime-host standard-trait accessor lint run: cargo clippy -p pantograph-runtime-host-contracts --all-targets -- -D clippy::should_implement_trait + - name: Run diagnostics ledger contract tests + run: cargo test -p pantograph-diagnostics-ledger - name: Run media conversion contract tests run: cargo test -p pantograph-media-conversion diff --git a/crates/pantograph-diagnostics-ledger/src/sqlite/event_sqlite.rs b/crates/pantograph-diagnostics-ledger/src/sqlite/event_sqlite.rs index 30e6c2d0d..528a0850d 100644 --- a/crates/pantograph-diagnostics-ledger/src/sqlite/event_sqlite.rs +++ b/crates/pantograph-diagnostics-ledger/src/sqlite/event_sqlite.rs @@ -182,22 +182,31 @@ pub(super) fn upsert_projection_state( update.validate()?; let tx = ledger.conn.transaction()?; let record = if update.status == ProjectionStatus::Failed { - write_projection_failure_state( + write_projection_state( &tx, - update.projection_name.as_str(), - update.projection_version, - update.last_applied_event_seq, - update.rebuilt_at_ms, - update - .last_error - .as_deref() - .expect("failed projection state has an error"), - update - .last_error_at_ms - .expect("failed projection state has an error timestamp"), - update - .last_failed_event_seq - .expect("failed projection state has a failed event cursor"), + ProjectionStateWrite { + projection_name: update.projection_name.as_str(), + projection_version: update.projection_version, + last_applied_event_seq: update.last_applied_event_seq, + status: ProjectionStatus::Failed, + rebuilt_at_ms: update.rebuilt_at_ms, + last_error: Some( + update + .last_error + .as_deref() + .expect("failed projection state has an error"), + ), + last_error_at_ms: Some( + update + .last_error_at_ms + .expect("failed projection state has an error timestamp"), + ), + last_failed_event_seq: Some( + update + .last_failed_event_seq + .expect("failed projection state has a failed event cursor"), + ), + }, )? } else { write_projection_state( @@ -268,31 +277,6 @@ fn write_projection_success_state( ) } -fn write_projection_failure_state( - tx: &Transaction<'_>, - projection_name: &str, - projection_version: i64, - last_applied_event_seq: i64, - rebuilt_at_ms: Option, - last_error: &str, - last_error_at_ms: i64, - last_failed_event_seq: i64, -) -> Result { - write_projection_state( - tx, - ProjectionStateWrite { - projection_name, - projection_version, - last_applied_event_seq, - status: ProjectionStatus::Failed, - rebuilt_at_ms, - last_error: Some(last_error), - last_error_at_ms: Some(last_error_at_ms), - last_failed_event_seq: Some(last_failed_event_seq), - }, - ) -} - fn write_projection_state( tx: &Transaction<'_>, update: ProjectionStateWrite<'_>, @@ -1948,7 +1932,8 @@ fn inference_option_support_timeline_detail( ("requires backend support", counts.requires_backend_support), ] .into_iter() - .filter_map(|(label, count)| (count > 0).then(|| format!("{label} {count}"))) + .filter(|(_, count)| *count > 0) + .map(|(label, count)| format!("{label} {count}")) .collect::>(); (!parts.is_empty()).then(|| format!("option support {}", parts.join(", "))) @@ -4329,3 +4314,46 @@ where { rusqlite::Error::FromSqlConversionFailure(0, Type::Text, Box::new(error)) } + +#[cfg(test)] +mod projection_style_tests { + use super::inference_option_support_timeline_detail; + use crate::event::InferenceOptionSupportCounts; + + #[test] + fn option_support_detail_preserves_order_and_omits_zero_counts() { + let counts = InferenceOptionSupportCounts { + honored: 1, + mapped: 2, + defaulted: 3, + ignored: 4, + unsupported: 5, + rejected: 6, + conflict: 7, + model_unavailable: 8, + backend_unavailable: 9, + requires_model_support: 10, + requires_backend_support: 11, + }; + assert_eq!(inference_option_support_timeline_detail(&counts).as_deref(), Some( + "option support honored 1, mapped 2, defaulted 3, ignored 4, unsupported 5, rejected 6, conflict 7, model unavailable 8, backend unavailable 9, requires model support 10, requires backend support 11" + )); + let sparse = InferenceOptionSupportCounts { + mapped: 2, + rejected: 6, + ..InferenceOptionSupportCounts::default() + }; + assert_eq!( + inference_option_support_timeline_detail(&sparse).as_deref(), + Some("option support mapped 2, rejected 6") + ); + } + + #[test] + fn option_support_detail_is_absent_for_zero_counts() { + assert_eq!( + inference_option_support_timeline_detail(&InferenceOptionSupportCounts::default()), + None + ); + } +} diff --git a/crates/pantograph-diagnostics-ledger/src/tests.rs b/crates/pantograph-diagnostics-ledger/src/tests.rs index 4ea640319..72993acea 100644 --- a/crates/pantograph-diagnostics-ledger/src/tests.rs +++ b/crates/pantograph-diagnostics-ledger/src/tests.rs @@ -2222,6 +2222,10 @@ fn projection_state_persists_failure_health_and_success_clears_it() { }) .expect("failed projection state stores"); + assert_eq!(failed.projection_name, "scheduler_timeline"); + assert_eq!(failed.projection_version, 1); + assert_eq!(failed.last_applied_event_seq, 10); + assert_eq!(failed.rebuilt_at_ms, None); assert_eq!(failed.status, ProjectionStatus::Failed); assert_eq!( failed.last_error.as_deref(), @@ -2230,6 +2234,11 @@ fn projection_state_persists_failure_health_and_success_clears_it() { assert_eq!(failed.last_error_at_ms, Some(20)); assert_eq!(failed.last_failed_event_seq, Some(11)); + assert_eq!( + ledger.projection_state("scheduler_timeline").unwrap(), + Some(failed.clone()) + ); + let recovered = ledger .upsert_projection_state(ProjectionStateUpdate { projection_name: "scheduler_timeline".to_string(), diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-ledger-private-projection-input.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-ledger-private-projection-input.md new file mode 100644 index 000000000..1c436dc8a --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-ledger-private-projection-input.md @@ -0,0 +1,7 @@ +# Private ledger projection input and ordered timeline formatting + +Fresh PR #32 Clippy reports one private eight-argument projection writer and one bool::then filter_map in diagnostics-ledger. Reuse the existing ProjectionStateWrite at the sole failure-writer caller, preserving validation, transaction order, Failed status, all eight stored values, exact expect messages and the lower SQL writer. Remove only the redundant private wrapper. + +Replace the timeline filter_map with ordered filter/map. Exact all-category and sparse text tests preserve label order and omission of zero counts; an empty-count test preserves None. The existing failure-health/recovery test now checks every stable stored field and reads the failed state back before recovery. Hosted CI runs the complete ledger crate suite. The public payload enum layout remains a separate change. + +Whitespace and Rust syntax formatting are checked locally. Root source review accepted the exact four-file frozen tree ab465f369c064fc8c4d966e9fa99ae7f1ec1255f. Actual hosted Rust execution remains pending. From 443348201cf73424fc1705a6a9deb90f5021d5ef Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 06:26:46 -0700 Subject: [PATCH 15/25] refactor(diagnostics): box inference event payload --- CHANGELOG.md | 4 ++ .../src/event.rs | 2 +- .../src/tests.rs | 46 ++++++++++++++++++- .../src/node_execution_ledger.rs | 12 ++--- .../src/workflow/session_execution_api.rs | 4 +- ...6-10-03-ledger-inference-payload-layout.md | 7 +++ 6 files changed, 64 insertions(+), 11 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-ledger-inference-payload-layout.md diff --git a/CHANGELOG.md b/CHANGELOG.md index 67e487d99..94cbba6ec 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -51,3 +51,7 @@ The format is based on Keep a Changelog. ### Runtime-host borrowed accessor compatibility The seven validated runtime-host contract wrappers now implement standard `AsRef` in place of inherent `as_ref` methods. Borrowed values and lifetimes, wrapper types, validation and `into_inner` are unchanged. Ordinary method syntax, qualified calls and function pointers continue to resolve through the standard prelude. Rust callers using `no_implicit_prelude` must explicitly import `std::convert::AsRef`. The workspace crate remains unpublished (`publish = false`). + +### Diagnostic event inference payload construction + +`DiagnosticEventPayload::InferenceExecutionDiagnosticObserved` now stores `Box`. Rust callers constructing this public variant must wrap the existing raw payload in `Box::new`; owned pattern bindings now contain a box. Tagged serialized JSON, raw payload fields and validation remain unchanged. This adds one allocation for this event variant; no measured performance improvement is claimed. The crate remains `publish = false`. diff --git a/crates/pantograph-diagnostics-ledger/src/event.rs b/crates/pantograph-diagnostics-ledger/src/event.rs index a608c9c9d..90b92164c 100644 --- a/crates/pantograph-diagnostics-ledger/src/event.rs +++ b/crates/pantograph-diagnostics-ledger/src/event.rs @@ -272,7 +272,7 @@ pub enum DiagnosticEventPayload { RetentionPolicyChanged(RetentionPolicyChangedPayload), RuntimeCapabilityObserved(RuntimeCapabilityObservedPayload), NodeExecutionStatus(NodeExecutionStatusPayload), - InferenceExecutionDiagnosticObserved(InferenceExecutionDiagnosticObservedPayload), + InferenceExecutionDiagnosticObserved(Box), DiagnosticErrorOccurred(DiagnosticErrorOccurredPayload), } diff --git a/crates/pantograph-diagnostics-ledger/src/tests.rs b/crates/pantograph-diagnostics-ledger/src/tests.rs index 72993acea..2563b315c 100644 --- a/crates/pantograph-diagnostics-ledger/src/tests.rs +++ b/crates/pantograph-diagnostics-ledger/src/tests.rs @@ -1432,6 +1432,48 @@ fn diagnostic_event_ledger_projects_backend_and_task_on_node_status() { assert_eq!(nodes[0].selected_backend_key.as_deref(), Some("pytorch")); } +#[test] +fn inference_diagnostic_payload_preserves_exact_tagged_wire_and_ledger_round_trip() { + let expected = serde_json::json!({ + "payload_type": "inference_execution_diagnostic_observed", + "request_id": "req-wire", + "task_id": "text_generation", + "compatibility_issue_count": 0, + "option_support_counts": { + "honored": 0, "mapped": 0, "defaulted": 0, "ignored": 0, + "unsupported": 0, "rejected": 0, "conflict": 0, + "model_unavailable": 0, "backend_unavailable": 0, + "requires_model_support": 0, "requires_backend_support": 0 + } + }); + let payload: DiagnosticEventPayload = serde_json::from_value(expected.clone()).unwrap(); + assert_eq!(serde_json::to_value(&payload).unwrap(), expected); + let mut event = sample_inference_execution_diagnostic_event(); + event.payload = payload.clone(); + let mut ledger = SqliteDiagnosticsLedger::open_in_memory().unwrap(); + ledger + .append_diagnostic_event(event) + .expect("valid inference payload appends"); + let records = ledger.diagnostic_events_after(0, 10).unwrap(); + assert_eq!(records.len(), 1); + assert_eq!( + serde_json::from_str::(&records[0].payload_json).unwrap(), + expected + ); + assert_eq!( + serde_json::from_str::(&records[0].payload_json).unwrap(), + payload + ); +} + +#[test] +fn diagnostic_event_payload_keeps_large_inference_record_indirect() { + assert!( + std::mem::size_of::() + < std::mem::size_of::() + ); +} + #[test] fn diagnostic_event_ledger_appends_inference_execution_diagnostic_summary() { let mut ledger = SqliteDiagnosticsLedger::open_in_memory().expect("ledger opens"); @@ -6404,7 +6446,7 @@ fn sample_inference_execution_diagnostic_event() -> DiagnosticEventAppendRequest retention_class: DiagnosticEventRetentionClass::AuditMetadata, payload_ref: None, payload: DiagnosticEventPayload::InferenceExecutionDiagnosticObserved( - InferenceExecutionDiagnosticObservedPayload { + Box::new(InferenceExecutionDiagnosticObservedPayload { request_id: "req-a".to_string(), task_id: "text_generation".to_string(), lifecycle_phase: Some("task_validation".to_string()), @@ -6470,7 +6512,7 @@ fn sample_inference_execution_diagnostic_event() -> DiagnosticEventAppendRequest message: Some("not mapped by this backend boundary".to_string()), }, ], - }, + }), ), } } diff --git a/crates/pantograph-embedded-runtime/src/node_execution_ledger.rs b/crates/pantograph-embedded-runtime/src/node_execution_ledger.rs index 8a1979cde..499d35347 100644 --- a/crates/pantograph-embedded-runtime/src/node_execution_ledger.rs +++ b/crates/pantograph-embedded-runtime/src/node_execution_ledger.rs @@ -952,7 +952,7 @@ fn build_kv_cache_diagnostic_event_ledger_append_request( retention_class: DiagnosticEventRetentionClass::AuditMetadata, payload_ref: None, payload: DiagnosticEventPayload::InferenceExecutionDiagnosticObserved( - InferenceExecutionDiagnosticObservedPayload { + Box::new(InferenceExecutionDiagnosticObservedPayload { request_id: format!("{task_id}:kv_cache"), task_id: "kv_cache".to_string(), lifecycle_phase: Some("kv_cache".to_string()), @@ -981,7 +981,7 @@ fn build_kv_cache_diagnostic_event_ledger_append_request( .take(MAX_INFERENCE_OPTION_DIAGNOSTICS) .map(kv_cache_option_diagnostic_summary) .collect(), - }, + }), ), }) } @@ -1048,7 +1048,7 @@ fn build_runtime_settings_diagnostic_event_ledger_append_request( retention_class: DiagnosticEventRetentionClass::AuditMetadata, payload_ref: None, payload: DiagnosticEventPayload::InferenceExecutionDiagnosticObserved( - InferenceExecutionDiagnosticObservedPayload { + Box::new(InferenceExecutionDiagnosticObservedPayload { request_id: format!("{task_id}:runtime_settings"), task_id: "runtime_settings".to_string(), lifecycle_phase: Some("runtime_settings".to_string()), @@ -1075,7 +1075,7 @@ fn build_runtime_settings_diagnostic_event_ledger_append_request( compatibility_issues: Vec::new(), option_support_counts: InferenceOptionSupportCounts::default(), option_diagnostics: Vec::new(), - }, + }), ), }) } @@ -1147,7 +1147,7 @@ fn build_inference_diagnostic_event_ledger_append_request( retention_class: DiagnosticEventRetentionClass::AuditMetadata, payload_ref: None, payload: DiagnosticEventPayload::InferenceExecutionDiagnosticObserved( - InferenceExecutionDiagnosticObservedPayload { + Box::new(InferenceExecutionDiagnosticObservedPayload { request_id: event .request_id .clone() @@ -1196,7 +1196,7 @@ fn build_inference_diagnostic_event_ledger_append_request( .take(MAX_INFERENCE_OPTION_DIAGNOSTICS) .map(option_diagnostic_summary) .collect(), - }, + }), ), }) } diff --git a/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs b/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs index e8bd263f2..b636ed332 100644 --- a/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs +++ b/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs @@ -3103,7 +3103,7 @@ mod tests { retention_class: DiagnosticEventRetentionClass::AuditMetadata, payload_ref: None, payload: DiagnosticEventPayload::InferenceExecutionDiagnosticObserved( - pantograph_diagnostics_ledger::InferenceExecutionDiagnosticObservedPayload { + Box::new(pantograph_diagnostics_ledger::InferenceExecutionDiagnosticObservedPayload { request_id: "req-a".to_string(), task_id: "image_generation".to_string(), lifecycle_phase: Some("backend_execution".to_string()), @@ -3138,7 +3138,7 @@ mod tests { option_support_counts: pantograph_diagnostics_ledger::InferenceOptionSupportCounts::default(), option_diagnostics: Vec::new(), - }, + }), ), } } diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-ledger-inference-payload-layout.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-ledger-inference-payload-layout.md new file mode 100644 index 000000000..4d26752bc --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-ledger-inference-payload-layout.md @@ -0,0 +1,7 @@ +# Diagnostic event inference payload indirection + +Fresh PR #33 aggregate Clippy identifies InferenceExecutionDiagnosticObserved as the 872-byte largest DiagnosticEventPayload variant, with IoArtifactObserved next at 640 bytes. Box only the inference variant; preserve the raw payload type, tagged Serde representation, validation, event-kind mapping and projection readers. Migrate the five existing constructors: three embedded-runtime recorders, one ledger fixture and one workflow-service fixture. Other variants and Result types remain unchanged. + +An explicit minimal tagged JSON test deserializes the public enum, checks exact serialization and appends/reads it through the SQLite ledger. A layout test ensures the enum remains smaller than its raw inference payload without claiming throughput or allocation improvements. The inherited full ledger suite retains detailed inference append, projection, rejection and resource-rollup behavior coverage, and Headless qualifies the embedded-runtime callers. + +This is a public Rust construction and owned-pattern-binding change despite publish=false: callers wrap raw values in Box::new. It adds one allocation for this variant, with unchanged wire format and raw field semantics. Root source review accepted all six files at frozen tree e7a57e8d1bc9db3100c75dcb45ae9ef4dad8d67a. Hosted execution remains pending. No local Rust execution is claimed. From 3d8de843930154892fa21903bf57e4933d796b14 Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 06:41:49 -0700 Subject: [PATCH 16/25] refactor(diagnostics): box artifact event payload --- CHANGELOG.md | 4 ++ .../src/event.rs | 2 +- .../src/tests.rs | 44 ++++++++++++++++++- .../src/node_execution_ledger.rs | 4 +- .../src/workflow/session_execution_api.rs | 4 +- .../src/workflow/tests/diagnostics.rs | 4 +- ...26-10-03-ledger-artifact-payload-layout.md | 7 +++ 7 files changed, 60 insertions(+), 9 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-ledger-artifact-payload-layout.md diff --git a/CHANGELOG.md b/CHANGELOG.md index 94cbba6ec..2c8136067 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -55,3 +55,7 @@ The seven validated runtime-host contract wrappers now implement standard `AsRef ### Diagnostic event inference payload construction `DiagnosticEventPayload::InferenceExecutionDiagnosticObserved` now stores `Box`. Rust callers constructing this public variant must wrap the existing raw payload in `Box::new`; owned pattern bindings now contain a box. Tagged serialized JSON, raw payload fields and validation remain unchanged. This adds one allocation for this event variant; no measured performance improvement is claimed. The crate remains `publish = false`. + +### Diagnostic event artifact payload construction + +`DiagnosticEventPayload::IoArtifactObserved` now stores `Box`. Rust callers constructing the public variant use `Box::new`, and owned pattern bindings contain a box. Raw DTO fields, validation and tagged JSON stay unchanged; this adds one allocation for this variant without a measured performance claim. The crate remains `publish = false`. diff --git a/crates/pantograph-diagnostics-ledger/src/event.rs b/crates/pantograph-diagnostics-ledger/src/event.rs index 90b92164c..68c050673 100644 --- a/crates/pantograph-diagnostics-ledger/src/event.rs +++ b/crates/pantograph-diagnostics-ledger/src/event.rs @@ -266,7 +266,7 @@ pub enum DiagnosticEventPayload { RunStarted(RunStartedPayload), RunTerminal(RunTerminalPayload), RunSnapshotAccepted(RunSnapshotAcceptedPayload), - IoArtifactObserved(IoArtifactObservedPayload), + IoArtifactObserved(Box), RetentionArtifactStateChanged(RetentionArtifactStateChangedPayload), LibraryAssetAccessed(LibraryAssetAccessedPayload), RetentionPolicyChanged(RetentionPolicyChangedPayload), diff --git a/crates/pantograph-diagnostics-ledger/src/tests.rs b/crates/pantograph-diagnostics-ledger/src/tests.rs index 2563b315c..2948baa34 100644 --- a/crates/pantograph-diagnostics-ledger/src/tests.rs +++ b/crates/pantograph-diagnostics-ledger/src/tests.rs @@ -3051,6 +3051,46 @@ fn run_detail_projection_backfills_selected_runtime_from_node_status() { assert_eq!(record.selected_task_id.as_deref(), Some("text_generation")); } +#[test] +fn io_artifact_payload_preserves_exact_tagged_wire_and_ledger_round_trip() { + let expected = serde_json::json!({ + "payload_type": "io_artifact_observed", + "artifact_id": "artifact-wire", + "artifact_role": "node_output", + "producer_node_id": null, "producer_port_id": null, + "consumer_node_id": null, "consumer_port_id": null, + "media_type": null, "size_bytes": null, "content_hash": null, + "retention_state": null, "retention_reason": null + }); + let payload: DiagnosticEventPayload = serde_json::from_value(expected.clone()).unwrap(); + assert_eq!(serde_json::to_value(&payload).unwrap(), expected); + let mut event = + sample_io_artifact_event("run-wire", "node-wire", "node_output", "artifact-wire"); + event.payload = payload.clone(); + let mut ledger = SqliteDiagnosticsLedger::open_in_memory().unwrap(); + ledger + .append_diagnostic_event(event) + .expect("valid artifact payload appends"); + let records = ledger.diagnostic_events_after(0, 10).unwrap(); + assert_eq!(records.len(), 1); + assert_eq!( + serde_json::from_str::(&records[0].payload_json).unwrap(), + expected + ); + assert_eq!( + serde_json::from_str::(&records[0].payload_json).unwrap(), + payload + ); +} + +#[test] +fn diagnostic_event_payload_keeps_large_artifact_record_indirect() { + assert!( + std::mem::size_of::() + < std::mem::size_of::() + ); +} + #[test] fn io_artifact_projection_drains_artifact_events_incrementally() { let mut ledger = SqliteDiagnosticsLedger::open_in_memory().expect("ledger opens"); @@ -6262,7 +6302,7 @@ fn sample_io_artifact_event( privacy_class: DiagnosticEventPrivacyClass::SensitiveReference, retention_class: DiagnosticEventRetentionClass::PayloadReference, payload_ref: Some(format!("artifact://{artifact_id}")), - payload: DiagnosticEventPayload::IoArtifactObserved(IoArtifactObservedPayload { + payload: DiagnosticEventPayload::IoArtifactObserved(Box::new(IoArtifactObservedPayload { artifact_fact_id: None, payload_artifact_id: None, artifact_id: artifact_id.to_string(), @@ -6303,7 +6343,7 @@ fn sample_io_artifact_event( conversion_command_id: None, conversion_dependencies: Vec::new(), }), - }), + })), } } diff --git a/crates/pantograph-embedded-runtime/src/node_execution_ledger.rs b/crates/pantograph-embedded-runtime/src/node_execution_ledger.rs index 499d35347..183fa5f49 100644 --- a/crates/pantograph-embedded-runtime/src/node_execution_ledger.rs +++ b/crates/pantograph-embedded-runtime/src/node_execution_ledger.rs @@ -532,7 +532,7 @@ impl NodeExecutionWorkflowLedgerSink { privacy_class: artifact.privacy_class, retention_class: artifact.retention_class, payload_ref: artifact.payload_ref, - payload: DiagnosticEventPayload::IoArtifactObserved(IoArtifactObservedPayload { + payload: DiagnosticEventPayload::IoArtifactObserved(Box::new(IoArtifactObservedPayload { artifact_fact_id: Some(artifact.artifact_fact_id), payload_artifact_id: Some(artifact.payload_artifact_id), artifact_id: artifact.artifact_id, @@ -553,7 +553,7 @@ impl NodeExecutionWorkflowLedgerSink { read_handle: artifact.read_handle, stream_handle: None, format: artifact.format, - }), + })), }; self.workflow_service diff --git a/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs b/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs index b636ed332..e47fcaa84 100644 --- a/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs +++ b/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs @@ -1351,7 +1351,7 @@ impl WorkflowService { retention_class: metadata.retention_class, payload_ref: metadata.payload_ref.clone(), payload: DiagnosticEventPayload::IoArtifactObserved( - IoArtifactObservedPayload { + Box::new(IoArtifactObservedPayload { artifact_fact_id: Some(metadata.artifact_fact_id), payload_artifact_id: Some(metadata.payload_artifact_id), artifact_id: metadata.artifact_id, @@ -1388,7 +1388,7 @@ impl WorkflowService { read_handle: metadata.read_handle, stream_handle: metadata.stream_handle, format: metadata.format, - }, + }), ), }, ) diff --git a/crates/pantograph-workflow-service/src/workflow/tests/diagnostics.rs b/crates/pantograph-workflow-service/src/workflow/tests/diagnostics.rs index 19b542165..8c44f3122 100644 --- a/crates/pantograph-workflow-service/src/workflow/tests/diagnostics.rs +++ b/crates/pantograph-workflow-service/src/workflow/tests/diagnostics.rs @@ -2811,7 +2811,7 @@ fn sample_io_artifact_event( privacy_class: DiagnosticEventPrivacyClass::SensitiveReference, retention_class: DiagnosticEventRetentionClass::PayloadReference, payload_ref: Some(format!("artifact://{artifact_id}")), - payload: DiagnosticEventPayload::IoArtifactObserved(IoArtifactObservedPayload { + payload: DiagnosticEventPayload::IoArtifactObserved(Box::new(IoArtifactObservedPayload { artifact_fact_id: None, payload_artifact_id: None, artifact_id: artifact_id.to_string(), @@ -2836,7 +2836,7 @@ fn sample_io_artifact_event( read_handle: None, stream_handle: None, format: None, - }), + })), } } diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-ledger-artifact-payload-layout.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-ledger-artifact-payload-layout.md new file mode 100644 index 000000000..c6d254ec6 --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-ledger-artifact-payload-layout.md @@ -0,0 +1,7 @@ +# Diagnostic event artifact payload indirection + +Fresh PR #35 Clippy job 111211956000 identifies IoArtifactObserved as the remaining 640-byte largest DiagnosticEventPayload variant, compared with SchedulerRunAdmitted at 360 bytes. Box only this artifact variant. Preserve the raw DTO, validation, event-kind mapping, Serde tag and projection readers. Migrate the four existing constructors in embedded-runtime, workflow-service and their ledger/diagnostics fixtures. No other enum variant or Result type changes. + +Add explicit minimal tagged JSON serialization/deserialization and SQLite append/read round-trip coverage, plus an enum-size bound. Inherited full ledger tests retain artifact projection and retention behavior, while Headless qualifies production callers. Public Rust variant construction and owned pattern binding now involve Box; the documentation records the added allocation despite publish=false and makes no measured performance claim. + +Root approved this bounded follow-up from fresh evidence. Staged whitespace and new-test formatting are checked locally; root source review accepted all seven files at frozen tree 9271a602ebe593d0f45d29674ebc99ca7cb3a1cc; actual hosted execution remains pending. No local Rust execution is claimed. From 89ffaef6ee45832604d1d24eecd9e09f67214e61 Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 06:54:19 -0700 Subject: [PATCH 17/25] refactor(workflow): box private no-selection error payload --- .github/workflows/quality-gates.yml | 5 ++ .../src/scheduler/task_orchestrator.rs | 8 ++- .../src/scheduler/task_orchestrator_tests.rs | 61 +++++++++++++++++-- ...-03-orchestrator-selection-error-layout.md | 7 +++ 4 files changed, 73 insertions(+), 8 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-orchestrator-selection-error-layout.md diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml index 9963b1131..4a0aeda46 100644 --- a/.github/workflows/quality-gates.yml +++ b/.github/workflows/quality-gates.yml @@ -251,6 +251,11 @@ jobs: - name: Run workflow-nodes tests run: cargo test -p workflow-nodes --lib + - name: Run workflow scheduler orchestrator regression tests + run: | + cargo test -p pantograph-workflow-service --lib scheduler::task_orchestrator::tests:: -- --list > "$RUNNER_TEMP/orchestrator-tests.list" + python3 -c 'import pathlib, sys; names = [line for line in pathlib.Path(sys.argv[1]).read_text().splitlines() if line.startswith("scheduler::task_orchestrator::tests::") and line.endswith(": test")]; print("Orchestrator tests discovered:", len(names)); sys.exit(0 if names else 1)' "$RUNNER_TEMP/orchestrator-tests.list" + cargo test -p pantograph-workflow-service --lib scheduler::task_orchestrator::tests:: - name: Run workflow-service contract tests run: cargo test -p pantograph-workflow-service --test contract diff --git a/crates/pantograph-workflow-service/src/scheduler/task_orchestrator.rs b/crates/pantograph-workflow-service/src/scheduler/task_orchestrator.rs index 2f6217dc2..2f7835d9b 100644 --- a/crates/pantograph-workflow-service/src/scheduler/task_orchestrator.rs +++ b/crates/pantograph-workflow-service/src/scheduler/task_orchestrator.rs @@ -631,7 +631,7 @@ impl WorkflowSchedulerTaskOrchestrator { .await?; return Err( WorkflowSchedulerTaskOrchestratorError::RuntimeDispatchSelectionNoSelection( - selection, + Box::new(selection), ), ); } @@ -2115,7 +2115,9 @@ fn dispatch_selected_handoff_from_selection( ) -> Result { if selection.state != SchedulerDispatchSelectionState::Selected { return Err( - WorkflowSchedulerTaskOrchestratorError::RuntimeDispatchSelectionNoSelection(selection), + WorkflowSchedulerTaskOrchestratorError::RuntimeDispatchSelectionNoSelection( + Box::new(selection), + ), ); } let Some(dispatch_decision) = selection.dispatch_decision else { @@ -2812,7 +2814,7 @@ pub(crate) enum WorkflowSchedulerTaskOrchestratorError { #[error("runtime-host task input mapping failed")] RuntimeHostTaskInputMapping(WorkflowRuntimeHostTaskInputMappingError), #[error("scheduler dispatch selection did not select a runtime task")] - RuntimeDispatchSelectionNoSelection(SchedulerDispatchSelectionDecision), + RuntimeDispatchSelectionNoSelection(Box), #[error("reservation lifecycle contract validation failed: {0}")] ReservationLifecycleContract(ReservationLifecycleContractError), #[error("reservation lifecycle port failed: {0}")] diff --git a/crates/pantograph-workflow-service/src/scheduler/task_orchestrator_tests.rs b/crates/pantograph-workflow-service/src/scheduler/task_orchestrator_tests.rs index b71aae3ba..ec4be194c 100644 --- a/crates/pantograph-workflow-service/src/scheduler/task_orchestrator_tests.rs +++ b/crates/pantograph-workflow-service/src/scheduler/task_orchestrator_tests.rs @@ -689,6 +689,12 @@ async fn orchestrator_selects_scheduler_dispatch_before_runtime_host_port() { async fn orchestrator_does_not_dispatch_runtime_host_when_scheduler_selects_no_candidate() { let mut selection_request = dispatch_selection_request_fixture(); selection_request.candidates.clear(); + let expected_selection = select_scheduler_dispatch( + ValidatedSchedulerDispatchSelectionRequest::try_from(selection_request.clone()).unwrap(), + ) + .unwrap() + .into_inner(); + let expected_json = serde_json::to_value(&expected_selection).unwrap(); let port = Arc::new(RecordingRuntimeHostPort::with_response( runtime_host_response_fixture(), )); @@ -708,14 +714,59 @@ async fn orchestrator_does_not_dispatch_runtime_host_when_scheduler_selects_no_c .await .expect_err("no-selection diagnostics must stop before runtime host"); - assert!(matches!( - error, - WorkflowSchedulerTaskOrchestratorError::RuntimeDispatchSelectionNoSelection(selection) - if selection.state == SchedulerDispatchSelectionState::NoSelection - )); + assert_eq!( + error.to_string(), + "scheduler dispatch selection did not select a runtime task" + ); + let WorkflowSchedulerTaskOrchestratorError::RuntimeDispatchSelectionNoSelection(selection) = + error + else { + panic!("expected retained no-selection decision"); + }; + assert_eq!( + selection.state, + SchedulerDispatchSelectionState::NoSelection + ); + assert_eq!(*selection, expected_selection); + assert_eq!( + serde_json::to_value(selection.as_ref()).unwrap(), + expected_json + ); assert!(port.requests().is_empty()); } +#[test] +fn no_selection_handoff_error_preserves_the_complete_decision() { + let mut request = dispatch_selection_request_fixture(); + request.candidates.clear(); + let decision = select_scheduler_dispatch( + ValidatedSchedulerDispatchSelectionRequest::try_from(request).unwrap(), + ) + .unwrap() + .into_inner(); + let expected_json = serde_json::to_value(&decision).unwrap(); + let error = super::dispatch_selected_handoff_from_selection(decision.clone()).unwrap_err(); + assert_eq!( + error.to_string(), + "scheduler dispatch selection did not select a runtime task" + ); + let WorkflowSchedulerTaskOrchestratorError::RuntimeDispatchSelectionNoSelection(retained) = + error + else { + panic!("expected retained no-selection decision"); + }; + assert_eq!(*retained, decision); + assert_eq!( + serde_json::to_value(retained.as_ref()).unwrap(), + expected_json + ); +} + +#[test] +fn orchestrator_error_does_not_embed_the_large_selection_decision() { + assert!(std::mem::size_of::() <= 128); +} + #[tokio::test] async fn orchestrator_rejects_missing_runtime_input_before_runtime_host_port() { let selection_request = dispatch_selection_request_fixture(); diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-orchestrator-selection-error-layout.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-orchestrator-selection-error-layout.md new file mode 100644 index 000000000..f39d82bf6 --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-orchestrator-selection-error-layout.md @@ -0,0 +1,7 @@ +# Private orchestrator no-selection error payload + +Fresh PR #36 Clippy job 111214475174 clears diagnostics-ledger and reaches workflow-service: 134 library findings, including 89 large-error reports. The private WorkflowSchedulerTaskOrchestratorError embeds a 1,984-byte no-selection decision, versus its next 112-byte variant. Box only RuntimeDispatchSelectionNoSelection at its two constructors. Preserve the error Display text, raw scheduler decision type, complete diagnostics and JSON, borrowed readers, success paths and all Result signatures. + +The existing no-candidate integration test now checks complete decision/JSON equality and exact Display while retaining the assertion that no runtime-host request occurs. A direct handoff rejection test covers the other constructor, and a layout bound checks the private error remains at most 128 bytes. Existing successful selection and persisted terminal-diagnostic tests remain, with explicit nonzero discovery and the full orchestrator test module in CI. + +This adds one allocation on the no-selection error path without changing public contracts or claiming measured performance. Other style, public layout and arity findings remain separate and require fresh evidence. Root approved the bounded design. Staged whitespace and focused test formatting pass locally; root source review accepted all four files at frozen tree da720da95b3f4f6ac6bafb4cfb552937412d483d; hosted execution remains pending. From 25de27ead981379c1d2f122f9a4a7841dbec0b28 Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 07:04:47 -0700 Subject: [PATCH 18/25] refactor(workflow): preserve contracts while clearing style lints --- .github/workflows/quality-gates.yml | 5 +++ .../src/graph/connection_intent.rs | 20 ++++++++---- .../src/graph/inference_interface_request.rs | 23 +++++++++++-- .../graph/inference_validation_task_owner.rs | 10 ++---- .../graph/session_inference_validation_api.rs | 20 ++++++------ .../src/scheduler/task_orchestrator.rs | 6 ++-- .../src/workflow/attribution_api.rs | 4 +-- .../src/workflow/contracts.rs | 30 ++++++++++------- .../src/workflow/diagnostics_api.rs | 3 +- .../executable_validation_snapshot.rs | 1 - .../src/workflow/session_execution_api.rs | 11 ++----- .../src/workflow/session_queue_api.rs | 1 - .../src/workflow/session_runtime.rs | 3 +- .../src/workflow/task_graph.rs | 32 ++++++++++++++++--- .../2026-10-03-workflow-lint-basics.md | 7 ++++ 15 files changed, 113 insertions(+), 63 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-workflow-lint-basics.md diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml index 4a0aeda46..badccc11a 100644 --- a/.github/workflows/quality-gates.yml +++ b/.github/workflows/quality-gates.yml @@ -251,6 +251,11 @@ jobs: - name: Run workflow-nodes tests run: cargo test -p workflow-nodes --lib + - name: Run workflow style contract regressions + run: | + cargo test -p pantograph-workflow-service --lib lint_style_ -- --list > "$RUNNER_TEMP/workflow-style-tests.list" + python3 -c 'import pathlib, sys; names = [line for line in pathlib.Path(sys.argv[1]).read_text().splitlines() if "lint_style_regressions::lint_style_" in line and line.endswith(": test")]; print("Workflow style tests discovered:", len(names)); sys.exit(0 if names else 1)' "$RUNNER_TEMP/workflow-style-tests.list" + cargo test -p pantograph-workflow-service --lib lint_style_ - name: Run workflow scheduler orchestrator regression tests run: | cargo test -p pantograph-workflow-service --lib scheduler::task_orchestrator::tests:: -- --list > "$RUNNER_TEMP/orchestrator-tests.list" diff --git a/crates/pantograph-workflow-service/src/graph/connection_intent.rs b/crates/pantograph-workflow-service/src/graph/connection_intent.rs index 1b87a6c2a..379f8e3fe 100644 --- a/crates/pantograph-workflow-service/src/graph/connection_intent.rs +++ b/crates/pantograph-workflow-service/src/graph/connection_intent.rs @@ -35,7 +35,7 @@ struct ResolvedInputAnchor<'a> { port: PortDefinition, } -#[derive(Debug, Clone, Copy)] +#[derive(Debug, Clone, Copy, Default)] pub struct InferenceConnectionSurfaceView<'a> { surfaces: &'a [InferenceConnectionSurface], } @@ -53,12 +53,6 @@ impl<'a> InferenceConnectionSurfaceView<'a> { } } -impl Default for InferenceConnectionSurfaceView<'_> { - fn default() -> Self { - Self { surfaces: &[] } - } -} - fn is_static_llm_connection_input(port_id: &str) -> bool { matches!( port_id, @@ -984,3 +978,15 @@ mod tests { } } } + +#[cfg(test)] +mod lint_style_regressions { + use super::InferenceConnectionSurfaceView; + + #[test] + fn lint_style_default_connection_surface_has_no_current_node() { + assert!(InferenceConnectionSurfaceView::default() + .current_surface_for("node-a") + .is_none()); + } +} diff --git a/crates/pantograph-workflow-service/src/graph/inference_interface_request.rs b/crates/pantograph-workflow-service/src/graph/inference_interface_request.rs index 527db7027..40fe96ce1 100644 --- a/crates/pantograph-workflow-service/src/graph/inference_interface_request.rs +++ b/crates/pantograph-workflow-service/src/graph/inference_interface_request.rs @@ -282,9 +282,7 @@ fn parse_model_ref( value: Option<&Value>, diagnostics: &mut Vec, ) -> Option { - let Some(value) = value else { - return None; - }; + let value = value?; match serde_json::from_value::(value.clone()) { Ok(model_ref) => match model_ref.validate() { @@ -895,3 +893,22 @@ mod tests { }) } } + +#[cfg(test)] +mod lint_style_regressions { + use super::{parse_model_ref, InferenceInterfaceGraphResolutionDiagnosticCode}; + + #[test] + fn lint_style_absent_model_ref_does_not_emit_invalid_value_diagnostic() { + let mut diagnostics = Vec::new(); + assert!(parse_model_ref("node-a", None, &mut diagnostics).is_none()); + assert!(diagnostics.is_empty()); + let invalid = serde_json::json!(7); + assert!(parse_model_ref("node-a", Some(&invalid), &mut diagnostics).is_none()); + assert_eq!(diagnostics.len(), 1); + assert_eq!( + diagnostics[0].code, + InferenceInterfaceGraphResolutionDiagnosticCode::InvalidPumasModelRef + ); + } +} diff --git a/crates/pantograph-workflow-service/src/graph/inference_validation_task_owner.rs b/crates/pantograph-workflow-service/src/graph/inference_validation_task_owner.rs index 0fd4534c9..4917824dd 100644 --- a/crates/pantograph-workflow-service/src/graph/inference_validation_task_owner.rs +++ b/crates/pantograph-workflow-service/src/graph/inference_validation_task_owner.rs @@ -89,7 +89,7 @@ impl WorkflowGraphValidationTaskOwner { let mut state = session_handle.lock().await; state.touch(); state.canonicalize_graph(); - WorkflowGraphRevision::parse(&state.graph.compute_fingerprint()) + WorkflowGraphRevision::parse(state.graph.compute_fingerprint()) .map_err(|error| WorkflowServiceError::InvalidRequest(error.to_string())) }, ) @@ -183,12 +183,8 @@ impl WorkflowGraphValidationTaskOwner { let finished_keys = state .active .iter() - .filter_map(|(graph_session_id, record)| { - record - .handle - .is_finished() - .then(|| graph_session_id.clone()) - }) + .filter(|(_, record)| record.handle.is_finished()) + .map(|(graph_session_id, _)| graph_session_id.clone()) .collect::>(); finished_keys .into_iter() diff --git a/crates/pantograph-workflow-service/src/graph/session_inference_validation_api.rs b/crates/pantograph-workflow-service/src/graph/session_inference_validation_api.rs index aa31ecfd7..2007e700a 100644 --- a/crates/pantograph-workflow-service/src/graph/session_inference_validation_api.rs +++ b/crates/pantograph-workflow-service/src/graph/session_inference_validation_api.rs @@ -48,7 +48,7 @@ impl GraphSessionStore { state.touch(); state.canonicalize_graph(); let current_graph_revision = - WorkflowGraphRevision::parse(&state.graph.compute_fingerprint()) + WorkflowGraphRevision::parse(state.graph.compute_fingerprint()) .map_err(|error| WorkflowServiceError::InvalidRequest(error.to_string()))?; drop(state); @@ -73,7 +73,7 @@ impl GraphSessionStore { state.touch(); state.canonicalize_graph(); let current_graph_revision = - WorkflowGraphRevision::parse(&state.graph.compute_fingerprint()) + WorkflowGraphRevision::parse(state.graph.compute_fingerprint()) .map_err(|error| WorkflowServiceError::InvalidRequest(error.to_string()))?; drop(state); @@ -98,7 +98,7 @@ impl GraphSessionStore { state.touch(); state.canonicalize_graph(); let graph = state.graph.clone(); - let current_graph_revision = WorkflowGraphRevision::parse(&graph.compute_fingerprint()) + let current_graph_revision = WorkflowGraphRevision::parse(graph.compute_fingerprint()) .map_err(|error| WorkflowServiceError::InvalidRequest(error.to_string()))?; drop(state); @@ -178,7 +178,7 @@ impl GraphSessionStore { let mut state = handle.lock().await; state.touch(); state.canonicalize_graph(); - let graph_revision = WorkflowGraphRevision::parse(&state.graph.compute_fingerprint()) + let graph_revision = WorkflowGraphRevision::parse(state.graph.compute_fingerprint()) .map_err(|error| WorkflowServiceError::InvalidRequest(error.to_string()))?; drop(state); @@ -223,7 +223,7 @@ impl GraphSessionStore { state.touch(); state.canonicalize_graph(); let current_graph_revision = - WorkflowGraphRevision::parse(&state.graph.compute_fingerprint()) + WorkflowGraphRevision::parse(state.graph.compute_fingerprint()) .map_err(|error| WorkflowServiceError::InvalidRequest(error.to_string()))?; if current_graph_revision != request.graph_revision { return Err(WorkflowServiceError::InvalidRequest( @@ -307,7 +307,7 @@ impl GraphSessionStore { state.touch(); state.canonicalize_graph(); let graph = state.graph.clone(); - let current_graph_revision = WorkflowGraphRevision::parse(&graph.compute_fingerprint()) + let current_graph_revision = WorkflowGraphRevision::parse(graph.compute_fingerprint()) .map_err(|error| WorkflowServiceError::InvalidRequest(error.to_string()))?; drop(state); @@ -373,7 +373,7 @@ impl GraphSessionStore { state.touch(); state.canonicalize_graph(); let current_graph_revision = - WorkflowGraphRevision::parse(&state.graph.compute_fingerprint()) + WorkflowGraphRevision::parse(state.graph.compute_fingerprint()) .map_err(|error| WorkflowServiceError::InvalidRequest(error.to_string()))?; if current_graph_revision != validation_session.graph_revision { return Err(WorkflowServiceError::InvalidRequest( @@ -401,7 +401,7 @@ impl GraphSessionStore { state.touch(); state.canonicalize_graph(); let graph = state.graph.clone(); - let graph_revision = WorkflowGraphRevision::parse(&graph.compute_fingerprint()) + let graph_revision = WorkflowGraphRevision::parse(graph.compute_fingerprint()) .map_err(|error| WorkflowServiceError::InvalidRequest(error.to_string()))?; drop(state); @@ -444,7 +444,7 @@ impl GraphSessionStore { let mut state = handle.lock().await; state.touch(); state.canonicalize_graph(); - WorkflowGraphRevision::parse(&state.graph.compute_fingerprint()) + WorkflowGraphRevision::parse(state.graph.compute_fingerprint()) .map_err(|error| WorkflowServiceError::InvalidRequest(error.to_string())) } @@ -461,7 +461,7 @@ impl GraphSessionStore { state.touch(); state.canonicalize_graph(); let graph = state.graph.clone(); - let graph_revision = WorkflowGraphRevision::parse(&graph.compute_fingerprint()) + let graph_revision = WorkflowGraphRevision::parse(graph.compute_fingerprint()) .map_err(|error| WorkflowServiceError::InvalidRequest(error.to_string()))?; drop(state); diff --git a/crates/pantograph-workflow-service/src/scheduler/task_orchestrator.rs b/crates/pantograph-workflow-service/src/scheduler/task_orchestrator.rs index 2f7835d9b..47365ce4b 100644 --- a/crates/pantograph-workflow-service/src/scheduler/task_orchestrator.rs +++ b/crates/pantograph-workflow-service/src/scheduler/task_orchestrator.rs @@ -636,13 +636,13 @@ impl WorkflowSchedulerTaskOrchestrator { ); } let handoff = dispatch_selected_handoff_from_selection(selection)?; - let dispatch_decision = handoff.dispatch_decision.as_ref().ok_or_else(|| { + let dispatch_decision = handoff.dispatch_decision.as_ref().ok_or( WorkflowSchedulerTaskOrchestratorError::SchedulerContract( SchedulerContractError::MissingField { field: "dispatch_decision", }, - ) - })?; + ), + )?; let reservation_lease_id = dispatch_decision.reservation_lease_id.clone(); let candidate_id = selected_candidate_id(&selection_request, dispatch_decision); Ok(SelectedRuntimeTaskDispatch { diff --git a/crates/pantograph-workflow-service/src/workflow/attribution_api.rs b/crates/pantograph-workflow-service/src/workflow/attribution_api.rs index 26c9f626e..175aeaf57 100644 --- a/crates/pantograph-workflow-service/src/workflow/attribution_api.rs +++ b/crates/pantograph-workflow-service/src/workflow/attribution_api.rs @@ -216,7 +216,7 @@ impl WorkflowService { )); } validate_workflow_id(&request.workflow_id)?; - let graph_revision = WorkflowGraphRevision::parse(&request.graph.compute_fingerprint()) + let graph_revision = WorkflowGraphRevision::parse(request.graph.compute_fingerprint()) .map_err(|error| WorkflowServiceError::InvalidRequest(error.to_string()))?; if graph_revision != request @@ -258,7 +258,7 @@ impl WorkflowService { request.validation_session_id.clone(), ) .await?; - let graph_revision = WorkflowGraphRevision::parse(&graph.compute_fingerprint()) + let graph_revision = WorkflowGraphRevision::parse(graph.compute_fingerprint()) .map_err(|error| WorkflowServiceError::InvalidRequest(error.to_string()))?; if graph_revision != source.graph_revision { return Err(WorkflowServiceError::InvalidRequest( diff --git a/crates/pantograph-workflow-service/src/workflow/contracts.rs b/crates/pantograph-workflow-service/src/workflow/contracts.rs index 68872fb5b..eb4377d45 100644 --- a/crates/pantograph-workflow-service/src/workflow/contracts.rs +++ b/crates/pantograph-workflow-service/src/workflow/contracts.rs @@ -95,7 +95,7 @@ pub struct WorkflowCapabilityModel { } /// Host capability payload consumed by the service. -#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)] +#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Default)] #[serde(rename_all = "snake_case")] pub struct WorkflowRuntimeRequirements { #[serde(default, skip_serializing_if = "Vec::is_empty")] @@ -105,17 +105,6 @@ pub struct WorkflowRuntimeRequirements { pub required_extensions: Vec, } -impl Default for WorkflowRuntimeRequirements { - fn default() -> Self { - Self { - resource_estimates: Vec::new(), - required_models: Vec::new(), - required_backends: Vec::new(), - required_extensions: Vec::new(), - } - } -} - #[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq, Eq)] #[serde(rename_all = "snake_case")] pub enum WorkflowRuntimeInstallState { @@ -1432,3 +1421,20 @@ impl Default for WorkflowRunHandle { Self::new() } } + +#[cfg(test)] +mod lint_style_regressions { + use super::WorkflowRuntimeRequirements; + + #[test] + fn lint_style_runtime_requirements_default_preserves_empty_wire_fields() { + let value = WorkflowRuntimeRequirements::default(); + assert!(value.resource_estimates.is_empty()); + assert_eq!( + serde_json::to_value(value).unwrap(), + serde_json::json!({ + "required_models": [], "required_backends": [], "required_extensions": [] + }) + ); + } +} diff --git a/crates/pantograph-workflow-service/src/workflow/diagnostics_api.rs b/crates/pantograph-workflow-service/src/workflow/diagnostics_api.rs index 09230eef4..da37b6b57 100644 --- a/crates/pantograph-workflow-service/src/workflow/diagnostics_api.rs +++ b/crates/pantograph-workflow-service/src/workflow/diagnostics_api.rs @@ -654,8 +654,7 @@ impl WorkflowService { resource_observation: None, }), }, - ) - .map_err(WorkflowServiceError::from)?; + )?; repaired = increment_startup_repair_count(repaired)?; } } diff --git a/crates/pantograph-workflow-service/src/workflow/executable_validation_snapshot.rs b/crates/pantograph-workflow-service/src/workflow/executable_validation_snapshot.rs index 1dc4b8b5b..d873efe0a 100644 --- a/crates/pantograph-workflow-service/src/workflow/executable_validation_snapshot.rs +++ b/crates/pantograph-workflow-service/src/workflow/executable_validation_snapshot.rs @@ -51,7 +51,6 @@ const SNAPSHOT_ID_PREFIX: &str = "wfvalsnap_"; pub struct WorkflowExecutableValidationSnapshotId(String); impl WorkflowExecutableValidationSnapshotId { - #[must_use] pub fn generate() -> Self { Self(format!("{SNAPSHOT_ID_PREFIX}{}", Uuid::new_v4())) } diff --git a/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs b/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs index e47fcaa84..5e0216ef5 100644 --- a/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs +++ b/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs @@ -1032,7 +1032,6 @@ impl WorkflowService { }, ) .map(|_| ()) - .map_err(WorkflowServiceError::from) } fn record_library_model_access_events_if_configured( @@ -1085,8 +1084,7 @@ impl WorkflowService { }, ), }, - ) - .map_err(WorkflowServiceError::from)?; + )?; } Ok(()) } @@ -1157,7 +1155,6 @@ impl WorkflowService { }, ) .map(|_| ()) - .map_err(WorkflowServiceError::from) } fn record_scheduler_queue_placement_event_if_configured( @@ -1217,7 +1214,6 @@ impl WorkflowService { }, ) .map(|_| ()) - .map_err(WorkflowServiceError::from) } pub(super) fn record_run_started_event_if_configured( @@ -1277,7 +1273,6 @@ impl WorkflowService { }, ) .map(|_| ()) - .map_err(WorkflowServiceError::from) } pub(super) fn record_workflow_io_artifact_events_if_configured( @@ -1391,8 +1386,7 @@ impl WorkflowService { }), ), }, - ) - .map_err(WorkflowServiceError::from)?; + )?; } Ok(()) } @@ -1481,7 +1475,6 @@ impl WorkflowService { }, ) .map(|_| ()) - .map_err(WorkflowServiceError::from) } } diff --git a/crates/pantograph-workflow-service/src/workflow/session_queue_api.rs b/crates/pantograph-workflow-service/src/workflow/session_queue_api.rs index 6e9892461..596a4501c 100644 --- a/crates/pantograph-workflow-service/src/workflow/session_queue_api.rs +++ b/crates/pantograph-workflow-service/src/workflow/session_queue_api.rs @@ -726,7 +726,6 @@ impl WorkflowService { }, ) .map(|_| ()) - .map_err(WorkflowServiceError::from) } fn record_scheduler_estimate_update_for_queue_item_if_configured( diff --git a/crates/pantograph-workflow-service/src/workflow/session_runtime.rs b/crates/pantograph-workflow-service/src/workflow/session_runtime.rs index e99db3310..0e5aba03c 100644 --- a/crates/pantograph-workflow-service/src/workflow/session_runtime.rs +++ b/crates/pantograph-workflow-service/src/workflow/session_runtime.rs @@ -348,8 +348,7 @@ impl WorkflowService { }, ), }, - ) - .map_err(WorkflowServiceError::from)?; + )?; } Ok(()) } diff --git a/crates/pantograph-workflow-service/src/workflow/task_graph.rs b/crates/pantograph-workflow-service/src/workflow/task_graph.rs index 46411da8e..23c586a8c 100644 --- a/crates/pantograph-workflow-service/src/workflow/task_graph.rs +++ b/crates/pantograph-workflow-service/src/workflow/task_graph.rs @@ -254,10 +254,8 @@ fn dependency_task_ids(bindings: &[WorkflowSchedulerTaskInputBinding]) -> Vec>(); + let ids = dependency_task_ids(&bindings); + assert_eq!( + ids.iter().map(|id| id.as_str()).collect::>(), + vec!["z", "a", "b"] + ); + assert!(dependency_task_ids(&[]).is_empty()); + } +} diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-workflow-lint-basics.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-workflow-lint-basics.md new file mode 100644 index 000000000..d0cabf73b --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-workflow-lint-basics.md @@ -0,0 +1,7 @@ +# Workflow behavior-preserving lint basics + +Prepare only the 30 mechanical findings evidenced by PR #36: 13 unnecessary fingerprint borrows, 10 identity WorkflowServiceError conversions, two equivalent empty defaults, two ordered filter/map rewrites, one absent-Option early return, one eager literal error construction, and one redundant method must_use annotation where the type retains it. Preserve every signature, error value, branch order, iterator order and default wire field. Other conversions, payload layouts, enum naming, type aliases and seven high-arity methods stay outside this slice. + +Four focused regressions cover empty connection surfaces, exact default runtime requirements JSON, absent versus invalid model references and first-seen dependency deduplication order. Explicit nonzero discovery runs those tests; inherited full service/Headless tests retain cancellation, session recovery and dispatch behavior. Finished-task filtering keeps the same state lock, one is_finished check per entry and identical removal sequence. + +Root approved the bounded preparation. Root source review accepted the entire fifteen-file frozen tree 61cb018d85a9c49b0da80bd9175fc1bd919fd764. Fresh PR #37 Clippy job 111216602528 confirms all 30 findings remain at the same locations/categories; the aggregate fell from 134 to 47 findings after the separate private error change. Hosted execution of this mechanical slice remains pending. No local Rust execution is claimed. From b4637a4800ad80bc983137cec0a6e98d7d0e24f3 Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 07:13:16 -0700 Subject: [PATCH 19/25] refactor(workflow): box private request and completion payloads --- .github/workflows/quality-gates.yml | 7 +++ .../src/graph/inference_validation_state.rs | 43 ++++++++++++++++++- .../src/workflow/task_execution_worker.rs | 27 +++++++++--- ...6-10-03-workflow-private-payload-layout.md | 7 +++ 4 files changed, 75 insertions(+), 9 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-workflow-private-payload-layout.md diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml index badccc11a..6cdec4212 100644 --- a/.github/workflows/quality-gates.yml +++ b/.github/workflows/quality-gates.yml @@ -251,6 +251,13 @@ jobs: - name: Run workflow-nodes tests run: cargo test -p workflow-nodes --lib + - name: Run private workflow owner regression suites + run: | + for module in graph::inference_validation_state::tests:: workflow::task_execution_worker::tests::; do + cargo test -p pantograph-workflow-service --lib "$module" -- --list > "$RUNNER_TEMP/private-owner-tests.list" + python3 -c 'import pathlib, sys; names = [line for line in pathlib.Path(sys.argv[1]).read_text().splitlines() if line.startswith(sys.argv[2]) and line.endswith(": test")]; print(sys.argv[2], "tests discovered:", len(names)); sys.exit(0 if names else 1)' "$RUNNER_TEMP/private-owner-tests.list" "$module" + cargo test -p pantograph-workflow-service --lib "$module" + done - name: Run workflow style contract regressions run: | cargo test -p pantograph-workflow-service --lib lint_style_ -- --list > "$RUNNER_TEMP/workflow-style-tests.list" diff --git a/crates/pantograph-workflow-service/src/graph/inference_validation_state.rs b/crates/pantograph-workflow-service/src/graph/inference_validation_state.rs index c0759411e..061f55ae0 100644 --- a/crates/pantograph-workflow-service/src/graph/inference_validation_state.rs +++ b/crates/pantograph-workflow-service/src/graph/inference_validation_state.rs @@ -1158,7 +1158,7 @@ pub(crate) enum DependencyEnvironmentActionIntentStateResolution { Blocked(DependencyEnvironmentActionIntentResult), RequestReady { intent: DependencyEnvironmentActionIntent, - environment_request: ValidatedDependencyEnvironmentRequest, + environment_request: Box, }, } @@ -1846,7 +1846,7 @@ fn request_ready_dependency_environment_action_resolution( ) -> DependencyEnvironmentActionIntentStateResolution { DependencyEnvironmentActionIntentStateResolution::RequestReady { intent, - environment_request, + environment_request: Box::new(environment_request), } } @@ -1877,6 +1877,15 @@ mod tests { use super::*; + #[test] + fn action_resolution_keeps_validated_environment_request_indirect() { + assert!(std::mem::size_of::() <= 384); + assert!( + std::mem::size_of::() + < std::mem::size_of::() + ); + } + #[tokio::test] async fn action_intent_state_rejects_stale_graph_revision() { let store = CurrentInferenceValidationStateStore::new(); @@ -2284,6 +2293,36 @@ mod tests { DependencyEnvironmentActionIntentStatus::RequestReady ); assert!(result.diagnostics.is_empty()); + + let resolution = store + .resolve_dependency_environment_action_request(state_request_with_validation_session( + "graph-session-1", + "aaaaaaaaaaaaaaaa", + "aaaaaaaaaaaaaaaa", + "validation.session.1", + "dependency-node-1", + true, + )) + .await; + let DependencyEnvironmentActionIntentStateResolution::RequestReady { + intent, + environment_request, + } = resolution + else { + panic!("current proof must retain a ready validated request"); + }; + let raw = environment_request.as_request(); + assert_eq!(raw.action, intent.action); + assert_eq!(raw.planning_request.task_id.as_str(), "image_generation"); + assert_eq!( + raw.identity_key, + DependencyPlanningIdentityKey::from_planning_request(&raw.planning_request).unwrap() + ); + raw.validate() + .expect("retained environment request validates"); + let encoded = serde_json::to_value(raw).unwrap(); + let decoded = ValidatedDependencyEnvironmentRequest::try_from(encoded.clone()).unwrap(); + assert_eq!(serde_json::to_value(decoded.as_request()).unwrap(), encoded); } #[tokio::test] diff --git a/crates/pantograph-workflow-service/src/workflow/task_execution_worker.rs b/crates/pantograph-workflow-service/src/workflow/task_execution_worker.rs index 5a15d2c97..e632a73db 100644 --- a/crates/pantograph-workflow-service/src/workflow/task_execution_worker.rs +++ b/crates/pantograph-workflow-service/src/workflow/task_execution_worker.rs @@ -368,7 +368,7 @@ pub(super) enum WorkflowTaskExecutionWorkerShutdownReason { pub(super) enum WorkflowTaskExecutionWorkerOutcome { TaskTerminal(WorkflowTaskExecutionWorkerTerminalOutcome), TaskDeferred(WorkflowTaskExecutionWorkerDeferredOutcome), - RuntimeBranchCompleted(WorkflowTaskExecutionWorkerRuntimeBranchCompletedOutcome), + RuntimeBranchCompleted(Box), RuntimeBranchFailed(WorkflowTaskExecutionWorkerRuntimeBranchFailedOutcome), RuntimeBranchCancelled(WorkflowTaskExecutionWorkerRuntimeBranchCancelledOutcome), RuntimeBranchDeferred(WorkflowTaskExecutionWorkerRuntimeBranchDeferredOutcome), @@ -1124,12 +1124,12 @@ impl WorkflowTaskExecutionWorkerOutcome { response: WorkflowRunResponse, diagnostics: Vec, ) -> Self { - Self::RuntimeBranchCompleted(WorkflowTaskExecutionWorkerRuntimeBranchCompletedOutcome { + Self::RuntimeBranchCompleted(Box::new(WorkflowTaskExecutionWorkerRuntimeBranchCompletedOutcome { session_id: command.session_id.clone(), workflow_run_id: command.workflow_run_id.clone(), response, diagnostics, - }) + })) } pub(super) fn runtime_branch_failed( @@ -1370,12 +1370,12 @@ async fn claim_and_execute_runtime_branch_event( .await { Ok(response) => WorkflowTaskExecutionWorkerOutcome::RuntimeBranchCompleted( - WorkflowTaskExecutionWorkerRuntimeBranchCompletedOutcome { + Box::new(WorkflowTaskExecutionWorkerRuntimeBranchCompletedOutcome { session_id: command.session_id.clone(), workflow_run_id: command.workflow_run_id.clone(), response, diagnostics: Vec::new(), - }, + }), ), Err(error) => WorkflowTaskExecutionWorkerOutcome::runtime_branch_failed( command, @@ -1886,12 +1886,12 @@ fn runtime_branch_batch_member_completion( unix_timestamp_ms(), proof) { Ok(_record) => match member_outcome.completed_response { Some(response) => WorkflowTaskExecutionWorkerOutcome::RuntimeBranchCompleted( - WorkflowTaskExecutionWorkerRuntimeBranchCompletedOutcome { + Box::new(WorkflowTaskExecutionWorkerRuntimeBranchCompletedOutcome { session_id: member_outcome.session_id.clone(), workflow_run_id: member_outcome.workflow_run_id.clone(), response, diagnostics, - }, + }), ), None => WorkflowTaskExecutionWorkerOutcome::RuntimeBranchFailed( WorkflowTaskExecutionWorkerRuntimeBranchFailedOutcome { @@ -4353,6 +4353,11 @@ mod tests { assert_eq!(outcome.diagnostics, vec![diagnostic]); } + #[test] + fn worker_outcome_keeps_completed_response_indirect() { + assert!(std::mem::size_of::() <= 128); + } + #[test] fn runtime_branch_completed_outcome_preserves_run_scope_and_response() { let command = runtime_branch_command(); @@ -4379,6 +4384,14 @@ mod tests { assert_eq!(outcome.session_id, command.session_id); assert_eq!(outcome.workflow_run_id, command.workflow_run_id); assert_eq!(outcome.response, response); + assert_eq!( + serde_json::to_value(&outcome.response).unwrap(), + serde_json::json!({ + "workflow_run_id": command.workflow_run_id, + "outputs": [], + "timing_ms": 42 + }) + ); assert_eq!(outcome.diagnostics, vec![diagnostic]); } diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-workflow-private-payload-layout.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-workflow-private-payload-layout.md new file mode 100644 index 000000000..bc0ce099b --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-workflow-private-payload-layout.md @@ -0,0 +1,7 @@ +# Private validation request and worker completion payloads + +Fresh PR #37 Clippy retains an 896-byte private RequestReady branch and six large-error reports caused by a 144-byte private RuntimeBranchCompleted outcome. Box only RequestReady.environment_request and the completed worker outcome. The constructor inventory is one validated-request helper and three completed-outcome constructors. The graph session still borrows the validated request for the dependency service; worker response extraction and all Result signatures remain unchanged. Public contracts and the public Ready inference projection are excluded. + +The real ready-proof regression now checks the retained request task, canonical identity, validation and full JSON round trip. Existing blocked/missing/stale proof tests remain. The worker success test retains complete response/scope/diagnostic equality and adds exact response JSON; existing cancellation, unavailable and shutdown behaviors remain. Two private layout bounds provide source-layout evidence, not measured performance claims. CI explicitly discovers and runs both complete owner test modules with nonzero guards. + +Root approved this bounded design. Whitespace and focused test syntax formatting are checked locally; root source review accepted all four files at frozen tree c2a6e2439f8df3a7e4d00461cf36175fbbe19ff4; hosted execution remains pending. No local Rust execution is claimed. From 09ae3182b3efa0e7921763122b25d0f0fc2ada11 Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 07:22:35 -0700 Subject: [PATCH 20/25] refactor(workflow): box ready inference projection payload --- .github/workflows/quality-gates.yml | 7 +++ CHANGELOG.md | 4 ++ .../src/graph/inference_validation_state.rs | 4 +- .../executable_validation_snapshot.rs | 4 +- .../src/workflow/task_graph.rs | 2 +- .../workflow/tests/task_binding_resolution.rs | 4 +- .../src/workflow/tests/task_graph.rs | 50 ++++++++++++++++++- ...10-03-ready-inference-projection-layout.md | 7 +++ 8 files changed, 73 insertions(+), 9 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-ready-inference-projection-layout.md diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml index 6cdec4212..a28eea8d2 100644 --- a/.github/workflows/quality-gates.yml +++ b/.github/workflows/quality-gates.yml @@ -251,6 +251,13 @@ jobs: - name: Run workflow-nodes tests run: cargo test -p workflow-nodes --lib + - name: Run workflow inference projection regression suites + run: | + for module in workflow::tests::task_graph:: workflow::tests::task_binding_resolution:: workflow::executable_validation_snapshot::tests::; do + cargo test -p pantograph-workflow-service --lib "$module" -- --list > "$RUNNER_TEMP/projection-tests.list" + python3 -c 'import pathlib, sys; names = [line for line in pathlib.Path(sys.argv[1]).read_text().splitlines() if line.startswith(sys.argv[2]) and line.endswith(": test")]; print(sys.argv[2], "tests discovered:", len(names)); sys.exit(0 if names else 1)' "$RUNNER_TEMP/projection-tests.list" "$module" + cargo test -p pantograph-workflow-service --lib "$module" + done - name: Run private workflow owner regression suites run: | for module in graph::inference_validation_state::tests:: workflow::task_execution_worker::tests::; do diff --git a/CHANGELOG.md b/CHANGELOG.md index 2c8136067..ebc94a76a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -59,3 +59,7 @@ The seven validated runtime-host contract wrappers now implement standard `AsRef ### Diagnostic event artifact payload construction `DiagnosticEventPayload::IoArtifactObserved` now stores `Box`. Rust callers constructing the public variant use `Box::new`, and owned pattern bindings contain a box. Raw DTO fields, validation and tagged JSON stay unchanged; this adds one allocation for this variant without a measured performance claim. The crate remains `publish = false`. + +### Workflow ready inference projection construction + +`WorkflowSchedulerInferenceTaskProjection::Ready` now holds `Box`. Rust constructors use `Box::new`, and owned pattern bindings contain a box; the raw record, equality, borrowed readers and downstream scheduler intent serialization remain unchanged. The projection enum itself has no Serde implementation. This adds one allocation for ready projections without a measured performance claim; the crate remains `publish = false`. diff --git a/crates/pantograph-workflow-service/src/graph/inference_validation_state.rs b/crates/pantograph-workflow-service/src/graph/inference_validation_state.rs index 061f55ae0..000e8c783 100644 --- a/crates/pantograph-workflow-service/src/graph/inference_validation_state.rs +++ b/crates/pantograph-workflow-service/src/graph/inference_validation_state.rs @@ -1468,7 +1468,7 @@ impl CurrentInferenceValidationNodeRecord { }, )?; return Ok(WorkflowSchedulerInferenceTaskProjection::Ready( - WorkflowSchedulerReadyInferenceTaskProjection { + Box::new(WorkflowSchedulerReadyInferenceTaskProjection { node_id: pantograph_scheduler::SchedulerNodeId::parse(self.node_id.as_str()) .map_err(|error| { CurrentInferenceSchedulerProjectionError::IncompleteNodeState { @@ -1532,7 +1532,7 @@ impl CurrentInferenceValidationNodeRecord { .dependency_override_fingerprint .clone(), }, - }, + }), )); } diff --git a/crates/pantograph-workflow-service/src/workflow/executable_validation_snapshot.rs b/crates/pantograph-workflow-service/src/workflow/executable_validation_snapshot.rs index d873efe0a..7c35fa013 100644 --- a/crates/pantograph-workflow-service/src/workflow/executable_validation_snapshot.rs +++ b/crates/pantograph-workflow-service/src/workflow/executable_validation_snapshot.rs @@ -492,7 +492,7 @@ impl WorkflowExecutableValidationSnapshotNode { })?; Ok(WorkflowSchedulerInferenceTaskProjection::Ready( - WorkflowSchedulerReadyInferenceTaskProjection { + Box::new(WorkflowSchedulerReadyInferenceTaskProjection { node_id: scheduler_node_id, descriptor_fingerprint: self.descriptor_fingerprint.clone(), task_type, @@ -504,7 +504,7 @@ impl WorkflowExecutableValidationSnapshotNode { dependency_readiness_source: workflow_scheduler_dependency_readiness_source( snapshot, self, )?, - }, + }), )) } } diff --git a/crates/pantograph-workflow-service/src/workflow/task_graph.rs b/crates/pantograph-workflow-service/src/workflow/task_graph.rs index 23c586a8c..89b9e71f5 100644 --- a/crates/pantograph-workflow-service/src/workflow/task_graph.rs +++ b/crates/pantograph-workflow-service/src/workflow/task_graph.rs @@ -64,7 +64,7 @@ impl WorkflowSchedulerInferenceTaskProjections { #[derive(Debug, Clone, PartialEq, Eq)] pub enum WorkflowSchedulerInferenceTaskProjection { - Ready(WorkflowSchedulerReadyInferenceTaskProjection), + Ready(Box), Blocked(WorkflowSchedulerBlockedInferenceTaskProjection), } diff --git a/crates/pantograph-workflow-service/src/workflow/tests/task_binding_resolution.rs b/crates/pantograph-workflow-service/src/workflow/tests/task_binding_resolution.rs index ea7fa28d8..186d08109 100644 --- a/crates/pantograph-workflow-service/src/workflow/tests/task_binding_resolution.rs +++ b/crates/pantograph-workflow-service/src/workflow/tests/task_binding_resolution.rs @@ -34,7 +34,7 @@ fn workflow_run_id() -> WorkflowRunId { fn inference_projection() -> WorkflowSchedulerInferenceTaskProjections { WorkflowSchedulerInferenceTaskProjections::from_records(vec![ WorkflowSchedulerInferenceTaskProjection::Ready( - WorkflowSchedulerReadyInferenceTaskProjection { + Box::new(WorkflowSchedulerReadyInferenceTaskProjection { node_id: SchedulerNodeId::parse("infer").expect("node id"), descriptor_fingerprint: InferenceInterfaceFingerprint::parse("iface.binding.v1") .expect("fingerprint"), @@ -50,7 +50,7 @@ fn inference_projection() -> WorkflowSchedulerInferenceTaskProjections { trait_settings: Vec::new(), estimate_hints: resource_estimate_hints(), dependency_readiness_source: dependency_readiness_source("iface.binding.v1"), - }, + }), ), ]) .expect("projection") diff --git a/crates/pantograph-workflow-service/src/workflow/tests/task_graph.rs b/crates/pantograph-workflow-service/src/workflow/tests/task_graph.rs index 9660df775..ec9a18eaa 100644 --- a/crates/pantograph-workflow-service/src/workflow/tests/task_graph.rs +++ b/crates/pantograph-workflow-service/src/workflow/tests/task_graph.rs @@ -45,7 +45,7 @@ fn ready_inference_projection( ) -> WorkflowSchedulerInferenceTaskProjections { WorkflowSchedulerInferenceTaskProjections::from_records(vec![ WorkflowSchedulerInferenceTaskProjection::Ready( - WorkflowSchedulerReadyInferenceTaskProjection { + Box::new(WorkflowSchedulerReadyInferenceTaskProjection { node_id: SchedulerNodeId::parse("infer").expect("node id"), descriptor_fingerprint: InferenceInterfaceFingerprint::parse("iface.test.v1") .expect("fingerprint"), @@ -70,7 +70,7 @@ fn ready_inference_projection( }], estimate_hints, dependency_readiness_source: dependency_readiness_source("iface.test.v1"), - }, + }), ), ]) .expect("projection") @@ -200,6 +200,26 @@ fn graph_with_inline_inference_ref() -> WorkflowGraph { } } +#[test] +fn ready_projection_preserves_owned_raw_record_and_compact_layout() { + let projections = inference_projection(); + let projection = projections + .get(&SchedulerNodeId::parse("infer").unwrap()) + .unwrap(); + let WorkflowSchedulerInferenceTaskProjection::Ready(borrowed) = projection else { + panic!("expected ready projection"); + }; + let WorkflowSchedulerInferenceTaskProjection::Ready(owned) = projection.clone() else { + panic!("expected cloned ready projection"); + }; + let raw: WorkflowSchedulerReadyInferenceTaskProjection = *owned; + assert_eq!(&raw, borrowed.as_ref()); + assert!( + std::mem::size_of::() + < std::mem::size_of::() + ); +} + #[test] fn scheduler_task_graph_projects_path_free_inference_intent() { let graph = workflow_scheduler_task_graph_with_inference_projections( @@ -295,6 +315,32 @@ fn scheduler_task_graph_projects_path_free_inference_intent() { SchedulerTraitValue::String("euler_discrete".to_string()) ); + let expected_intent = json!({ + "contract_version": 1, + "workflow_id": "workflow-task-graph", + "workflow_run_id": "run-task-graph", + "node_id": "infer", + "task_id": "infer", + "task_type": "image_generation", + "model_ref": { + "model_id": "image/example/tiny-diffusion", + "revision": "main", + "selected_artifact_id": "diffusers-bundle" + }, + "constraints": { "requested_runtime_id": "pytorch", "requested_device_id": "cuda:0" }, + "trait_settings": [{ "trait_id": "denoising_scheduler", "value": { "kind": "string", "value": "euler_discrete" } }], + "estimate_hints": [ + { "kind": "peak_ram_bytes", "value": 2_147_483_648_u64 }, + { "kind": "peak_vram_bytes", "value": 4_294_967_296_u64 } + ] + }); + assert_eq!(serde_json::to_value(intent).unwrap(), expected_intent); + assert_eq!( + serde_json::from_value::(expected_intent) + .unwrap(), + *intent + ); + let encoded = serde_json::to_string(&graph).expect("encode task graph"); assert!(!encoded.contains("model_path")); assert!(!encoded.contains("/tmp/legacy-model")); diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-ready-inference-projection-layout.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-ready-inference-projection-layout.md new file mode 100644 index 000000000..7e0d24373 --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-ready-inference-projection-layout.md @@ -0,0 +1,7 @@ +# Public ready inference projection payload + +Fresh PR #38 Clippy retains the public Ready projection at 528 bytes versus the Blocked record at 80. Box only Ready and migrate four existing constructors: current validation state, executable snapshot projection and two workflow fixture helpers. Raw fields, equality, borrowed readers, freshness/admission checks and blocked behavior remain unchanged. The enum has no Serde implementation; the downstream scheduler intent wire contract is the serialization boundary to preserve. + +Tests cover owned raw-record equality and layout, plus exact downstream intent JSON and deserialization while retaining existing path-free projection assertions. CI explicitly discovers and runs task-graph, task-binding and executable-snapshot modules; inherited validation-owner and Headless tests retain ready/stale/missing-estimate/blocked behavior and the embedded-runtime consumer. + +This is a public Rust construction/owned-binding change despite publish=false, documented with the extra ready-record allocation and no measured performance claim. Root approved the bounded design. Root source review accepted all eight files at frozen tree 42349ac0a39dd876e5fca06ca4b285f3cc6eaca8. Hosted execution remains pending; no local Rust execution is claimed. From 0c085394d1f6864cc712b1e934c1c22d41d57f1a Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 07:37:35 -0700 Subject: [PATCH 21/25] refactor(workflow): group terminal diagnostic inputs --- .github/workflows/quality-gates.yml | 5 + .../runtime_branch_batch_execution.rs | 27 ++-- .../runtime_branch_run_finalization.rs | 148 ++++++++++++++---- .../src/workflow/session_scheduler_runner.rs | 107 +++++++------ .../2026-10-03-terminal-diagnostic-input.md | 9 ++ 5 files changed, 198 insertions(+), 98 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-terminal-diagnostic-input.md diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml index a28eea8d2..93fcf95c6 100644 --- a/.github/workflows/quality-gates.yml +++ b/.github/workflows/quality-gates.yml @@ -251,6 +251,11 @@ jobs: - name: Run workflow-nodes tests run: cargo test -p workflow-nodes --lib + - name: Run terminal diagnostic regression suite + run: | + cargo test -p pantograph-workflow-service --lib workflow::runtime_branch_run_finalization::tests:: -- --list > "$RUNNER_TEMP/terminal-diagnostic-tests.list" + python3 -c 'import pathlib, sys; names = [line for line in pathlib.Path(sys.argv[1]).read_text().splitlines() if line.startswith("workflow::runtime_branch_run_finalization::tests::") and line.endswith(": test")]; print("Terminal diagnostic tests discovered:", len(names)); sys.exit(0 if names else 1)' "$RUNNER_TEMP/terminal-diagnostic-tests.list" + cargo test -p pantograph-workflow-service --lib workflow::runtime_branch_run_finalization::tests:: - name: Run workflow inference projection regression suites run: | for module in workflow::tests::task_graph:: workflow::tests::task_binding_resolution:: workflow::executable_validation_snapshot::tests::; do diff --git a/crates/pantograph-workflow-service/src/workflow/runtime_branch_batch_execution.rs b/crates/pantograph-workflow-service/src/workflow/runtime_branch_batch_execution.rs index bb2cf341f..be0d264c8 100644 --- a/crates/pantograph-workflow-service/src/workflow/runtime_branch_batch_execution.rs +++ b/crates/pantograph-workflow-service/src/workflow/runtime_branch_batch_execution.rs @@ -24,6 +24,7 @@ use pantograph_scheduler::{ use super::runtime_branch_run_finalization::{ completed_scheduler_run_response, record_scheduler_task_attempt_terminal, + WorkflowSchedulerTaskAttemptTerminalInput, }; use super::runtime_dispatch_assignment::{ WorkflowRuntimeDispatchAssignmentBatchClaim, @@ -439,18 +440,20 @@ where })?; record_scheduler_task_attempt_terminal( service, - application.started_batch_member.started().task(), - application - .started_batch_member - .started() - .attempt_id() - .as_str(), - application.started_batch_member.started().started_at_ms(), - transition, - reason, - error_summary, - Some(application.started_batch_member.selected_dispatch()), - Some(terminal_mutation), + WorkflowSchedulerTaskAttemptTerminalInput { + task: application.started_batch_member.started().task(), + attempt_id: application + .started_batch_member + .started() + .attempt_id() + .as_str(), + started_at_ms: application.started_batch_member.started().started_at_ms(), + transition, + reason, + error_summary, + selected_dispatch: Some(application.started_batch_member.selected_dispatch()), + terminal_mutation: Some(terminal_mutation), + }, ) .map_err(|error| { WorkflowRuntimeBranchBatchExecutionFailure::active_run_member( diff --git a/crates/pantograph-workflow-service/src/workflow/runtime_branch_run_finalization.rs b/crates/pantograph-workflow-service/src/workflow/runtime_branch_run_finalization.rs index dfcc18b15..0289ed26b 100644 --- a/crates/pantograph-workflow-service/src/workflow/runtime_branch_run_finalization.rs +++ b/crates/pantograph-workflow-service/src/workflow/runtime_branch_run_finalization.rs @@ -37,6 +37,18 @@ pub(super) struct WorkflowSchedulerTaskAttemptDiagnosticAttribution { pub(super) bucket_id: Option, } +#[derive(Debug)] +pub(super) struct WorkflowSchedulerTaskAttemptTerminalInput<'a> { + pub(super) task: &'a WorkflowSchedulerTask, + pub(super) attempt_id: &'a str, + pub(super) started_at_ms: u64, + pub(super) transition: SchedulerTaskAttemptLifecycleTransition, + pub(super) reason: &'a str, + pub(super) error_summary: Option, + pub(super) selected_dispatch: Option<&'a SelectedRuntimeTaskDispatch>, + pub(super) terminal_mutation: Option<&'a WorkflowSchedulerTaskTerminalMutation>, +} + #[derive(Debug)] pub(super) struct WorkflowSchedulerTaskAttemptTerminalDiagnosticRequest<'a> { pub(super) task: &'a WorkflowSchedulerTask, @@ -142,14 +154,16 @@ pub(super) async fn finalize_started_runtime_task_dispatch( scheduler_task_attempt_terminal_transition_from_result(&result); record_scheduler_task_attempt_terminal( service, - started_runtime_task.task(), - started_runtime_task.attempt_id().as_str(), - started_runtime_task.started_at_ms(), - transition, - reason, - error_summary, - Some(selected_dispatch), - Some(&terminal_mutation), + WorkflowSchedulerTaskAttemptTerminalInput { + task: started_runtime_task.task(), + attempt_id: started_runtime_task.attempt_id().as_str(), + started_at_ms: started_runtime_task.started_at_ms(), + transition, + reason, + error_summary, + selected_dispatch: Some(selected_dispatch), + terminal_mutation: Some(&terminal_mutation), + }, )?; service .scheduler_task_orchestrator @@ -191,14 +205,16 @@ pub(super) async fn finalize_started_runtime_task_dispatch( }; record_scheduler_task_attempt_terminal( service, - started_runtime_task.task(), - started_runtime_task.attempt_id().as_str(), - started_runtime_task.started_at_ms(), - SchedulerTaskAttemptLifecycleTransition::Cancelled, - "scheduler runtime task cancellation observed", - Some(message.clone()), - Some(selected_dispatch), - Some(&terminal_mutation), + WorkflowSchedulerTaskAttemptTerminalInput { + task: started_runtime_task.task(), + attempt_id: started_runtime_task.attempt_id().as_str(), + started_at_ms: started_runtime_task.started_at_ms(), + transition: SchedulerTaskAttemptLifecycleTransition::Cancelled, + reason: "scheduler runtime task cancellation observed", + error_summary: Some(message.clone()), + selected_dispatch: Some(selected_dispatch), + terminal_mutation: Some(&terminal_mutation), + }, )?; service .scheduler_task_orchestrator @@ -234,14 +250,16 @@ pub(super) async fn finalize_started_runtime_task_dispatch( }; record_scheduler_task_attempt_terminal( service, - started_runtime_task.task(), - started_runtime_task.attempt_id().as_str(), - started_runtime_task.started_at_ms(), - SchedulerTaskAttemptLifecycleTransition::Failed, - "scheduler runtime task dispatch failed", - Some(error.to_string()), - Some(selected_dispatch), - Some(&terminal_mutation), + WorkflowSchedulerTaskAttemptTerminalInput { + task: started_runtime_task.task(), + attempt_id: started_runtime_task.attempt_id().as_str(), + started_at_ms: started_runtime_task.started_at_ms(), + transition: SchedulerTaskAttemptLifecycleTransition::Failed, + reason: "scheduler runtime task dispatch failed", + error_summary: Some(error.to_string()), + selected_dispatch: Some(selected_dispatch), + terminal_mutation: Some(&terminal_mutation), + }, )?; service .scheduler_task_orchestrator @@ -358,15 +376,18 @@ fn scheduler_task_attempt_terminal_diagnostic_event_at( pub(super) fn record_scheduler_task_attempt_terminal( service: &WorkflowService, - task: &WorkflowSchedulerTask, - attempt_id: &str, - started_at_ms: u64, - transition: SchedulerTaskAttemptLifecycleTransition, - reason: &str, - error_summary: Option, - selected_dispatch: Option<&SelectedRuntimeTaskDispatch>, - terminal_mutation: Option<&WorkflowSchedulerTaskTerminalMutation>, + input: WorkflowSchedulerTaskAttemptTerminalInput<'_>, ) -> Result<(), WorkflowServiceError> { + let WorkflowSchedulerTaskAttemptTerminalInput { + task, + attempt_id, + started_at_ms, + transition, + reason, + error_summary, + selected_dispatch, + terminal_mutation, + } = input; let attribution = scheduler_task_attempt_diagnostic_attribution(service, task.workflow_run_id.as_str())?; service.workflow_diagnostic_event_record(scheduler_task_attempt_terminal_diagnostic_event( @@ -541,6 +562,69 @@ mod tests { use super::*; + #[test] + fn terminal_input_records_success_failure_and_cancellation_scope() { + use pantograph_diagnostics_ledger::{DiagnosticsLedgerRepository, SqliteDiagnosticsLedger}; + + for transition in [ + SchedulerTaskAttemptLifecycleTransition::Completed, + SchedulerTaskAttemptLifecycleTransition::Failed, + SchedulerTaskAttemptLifecycleTransition::Cancelled, + ] { + let service = WorkflowService::new() + .with_diagnostics_ledger(SqliteDiagnosticsLedger::open_in_memory().unwrap()); + let task = runtime_task(); + let started_at_ms = unix_timestamp_ms(); + let error_summary = (transition != SchedulerTaskAttemptLifecycleTransition::Completed) + .then(|| "terminal fixture detail".to_string()); + record_scheduler_task_attempt_terminal( + &service, + WorkflowSchedulerTaskAttemptTerminalInput { + task: &task, + attempt_id: "attempt.terminal.input", + started_at_ms, + transition, + reason: "terminal input fixture", + error_summary: error_summary.clone(), + selected_dispatch: None, + terminal_mutation: None, + }, + ) + .expect("terminal input records event"); + let records = service + .diagnostics_ledger_guard() + .unwrap() + .diagnostic_events_after(0, 10) + .unwrap(); + assert_eq!(records.len(), 1); + let record = &records[0]; + assert_eq!( + record.workflow_run_id.as_ref().unwrap().as_str(), + task.workflow_run_id.as_str() + ); + assert_eq!(record.node_id.as_deref(), Some(task.node_id.as_str())); + assert!(record.client_id.is_none()); + assert!(record.client_session_id.is_none()); + assert!(record.bucket_id.is_none()); + let DiagnosticEventPayload::SchedulerTaskAttemptLifecycleChanged(payload) = + serde_json::from_str(&record.payload_json).unwrap() + else { + panic!("expected task-attempt terminal event"); + }; + assert_eq!(payload.scheduler_task_id, task.task_id.as_str()); + assert_eq!(payload.scheduler_attempt_id, "attempt.terminal.input"); + assert_eq!(payload.transition, transition); + assert_eq!(payload.started_at_ms, Some(started_at_ms as i64)); + assert_eq!( + payload.duration_ms, + Some((payload.ended_at_ms.unwrap() as u64).saturating_sub(started_at_ms)) + ); + assert_eq!(payload.reason.as_deref(), Some("terminal input fixture")); + assert_eq!(payload.error_summary, error_summary); + assert!(payload.reservation_id.is_none()); + } + } + #[test] fn terminal_diagnostic_event_preserves_scheduler_task_attempt_scope() { let task = runtime_task(); diff --git a/crates/pantograph-workflow-service/src/workflow/session_scheduler_runner.rs b/crates/pantograph-workflow-service/src/workflow/session_scheduler_runner.rs index 74fd40b9a..31456e718 100644 --- a/crates/pantograph-workflow-service/src/workflow/session_scheduler_runner.rs +++ b/crates/pantograph-workflow-service/src/workflow/session_scheduler_runner.rs @@ -24,7 +24,7 @@ use crate::scheduler::task_orchestrator::{ }; use crate::scheduler::{ WorkflowDependencyReadinessLifecycle, WorkflowDependencyReadinessLifecycleError, - WorkflowSchedulerRetryLifecycle, WorkflowSchedulerTaskTerminalMutation, + WorkflowSchedulerRetryLifecycle, }; use super::runtime_branch_run_finalization::{ @@ -33,6 +33,7 @@ use super::runtime_branch_run_finalization::{ WorkflowRuntimeTaskDispatchFinalizationOutcome, WorkflowSchedulerTaskAttemptDiagnosticAttribution, WorkflowSchedulerTaskAttemptTerminalDiagnosticRequest, + WorkflowSchedulerTaskAttemptTerminalInput, }; use super::{ WorkflowHost, WorkflowOutputTarget, WorkflowPortBinding, WorkflowRunResponse, @@ -513,17 +514,16 @@ impl<'a> WorkflowSchedulerSessionRunner<'a> { "scheduler non-runtime task completion failed: {error}" )) })?; - self.record_scheduler_task_attempt_terminal( - session_id, - started.task(), - started.attempt_id().as_str(), - started.started_at_ms(), - SchedulerTaskAttemptLifecycleTransition::Completed, - "scheduler task attempt completed", - None, - None, - None, - )?; + self.record_scheduler_task_attempt_terminal(WorkflowSchedulerTaskAttemptTerminalInput { + task: started.task(), + attempt_id: started.attempt_id().as_str(), + started_at_ms: started.started_at_ms(), + transition: SchedulerTaskAttemptLifecycleTransition::Completed, + reason: "scheduler task attempt completed", + error_summary: None, + selected_dispatch: None, + terminal_mutation: None, + })?; } Err( crate::scheduler::WorkflowSchedulerTaskOrchestratorError::NonRuntimeTaskAdapter( @@ -542,17 +542,16 @@ impl<'a> WorkflowSchedulerSessionRunner<'a> { &error, ); if failed.is_ok() { - self.record_scheduler_task_attempt_terminal( - session_id, - started.task(), - started.attempt_id().as_str(), - started.started_at_ms(), - SchedulerTaskAttemptLifecycleTransition::Failed, - "scheduler non-runtime task execution failed", - Some(error.to_string()), - None, - None, - )?; + self.record_scheduler_task_attempt_terminal(WorkflowSchedulerTaskAttemptTerminalInput { + task: started.task(), + attempt_id: started.attempt_id().as_str(), + started_at_ms: started.started_at_ms(), + transition: SchedulerTaskAttemptLifecycleTransition::Failed, + reason: "scheduler non-runtime task execution failed", + error_summary: Some(error.to_string()), + selected_dispatch: None, + terminal_mutation: None, + })?; } return Err(WorkflowServiceError::InvalidRequest(format!( "scheduler non-runtime task execution failed: {error}" @@ -886,17 +885,16 @@ impl<'a> WorkflowSchedulerSessionRunner<'a> { )) })? }; - self.record_scheduler_task_attempt_terminal( - session_id, - started_runtime_task.task(), - started_runtime_task.attempt_id().as_str(), - started_runtime_task.started_at_ms(), - SchedulerTaskAttemptLifecycleTransition::Failed, - "scheduler runtime dispatch selection failed", - Some(scheduler_error.to_string()), - None, - Some(&terminal_mutation), - )?; + self.record_scheduler_task_attempt_terminal(WorkflowSchedulerTaskAttemptTerminalInput { + task: started_runtime_task.task(), + attempt_id: started_runtime_task.attempt_id().as_str(), + started_at_ms: started_runtime_task.started_at_ms(), + transition: SchedulerTaskAttemptLifecycleTransition::Failed, + reason: "scheduler runtime dispatch selection failed", + error_summary: Some(scheduler_error.to_string()), + selected_dispatch: None, + terminal_mutation: Some(&terminal_mutation), + })?; } else { let terminal_mutation = { let mut store = self.service.session_store_guard()?; @@ -915,17 +913,16 @@ impl<'a> WorkflowSchedulerSessionRunner<'a> { )) })? }; - self.record_scheduler_task_attempt_terminal( - session_id, - started_runtime_task.task(), - started_runtime_task.attempt_id().as_str(), - started_runtime_task.started_at_ms(), - SchedulerTaskAttemptLifecycleTransition::Failed, - "scheduler runtime dispatch failed", - Some(scheduler_error.to_string()), - None, - Some(&terminal_mutation), - )?; + self.record_scheduler_task_attempt_terminal(WorkflowSchedulerTaskAttemptTerminalInput { + task: started_runtime_task.task(), + attempt_id: started_runtime_task.attempt_id().as_str(), + started_at_ms: started_runtime_task.started_at_ms(), + transition: SchedulerTaskAttemptLifecycleTransition::Failed, + reason: "scheduler runtime dispatch failed", + error_summary: Some(scheduler_error.to_string()), + selected_dispatch: None, + terminal_mutation: Some(&terminal_mutation), + })?; } return Err(WorkflowServiceError::CapabilityViolation(format!( "runtime scheduler dispatch selection failed closed for {count} runtime inference task(s): {scheduler_error}", @@ -1101,16 +1098,18 @@ impl<'a> WorkflowSchedulerSessionRunner<'a> { fn record_scheduler_task_attempt_terminal( &self, - _session_id: &str, - task: &WorkflowSchedulerTask, - attempt_id: &str, - started_at_ms: u64, - transition: SchedulerTaskAttemptLifecycleTransition, - reason: &str, - error_summary: Option, - selected_dispatch: Option<&SelectedRuntimeTaskDispatch>, - terminal_mutation: Option<&WorkflowSchedulerTaskTerminalMutation>, + input: WorkflowSchedulerTaskAttemptTerminalInput<'_>, ) -> Result<(), WorkflowServiceError> { + let WorkflowSchedulerTaskAttemptTerminalInput { + task, + attempt_id, + started_at_ms, + transition, + reason, + error_summary, + selected_dispatch, + terminal_mutation, + } = input; let attribution = self.scheduler_task_attempt_diagnostic_attribution(task.workflow_run_id.as_str())?; self.service.workflow_diagnostic_event_record( diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-terminal-diagnostic-input.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-terminal-diagnostic-input.md new file mode 100644 index 000000000..88a33f02f --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-terminal-diagnostic-input.md @@ -0,0 +1,9 @@ +# Shared private terminal diagnostic input + +Fresh PR #39 Clippy retains high-arity terminal diagnostic recorders in the session runner and runtime finalization owner. Group exactly eight existing values into private WorkflowSchedulerTaskAttemptTerminalInput: task, attempt_id, started_at_ms, transition, reason, owned error_summary, borrowed selected_dispatch and borrowed terminal_mutation. Migrate eight callers: four runner paths, three finalization paths and one batch path. + +Remove the runner's proven-unused _session_id argument. All four removed expressions were a plain session_id local with no side effects, and the parameter performed no validation. Both recorder bodies after destructuring remain byte-identical, including task.workflow_run_id attribution lookup and the unchanged lower event builder. No attribution-policy or public API change is made. + +A real in-memory ledger regression sends the new input through the recorder for success, failure and cancellation, verifying run/node/task/attempt scope, absent attribution, timing consistency, reason and original error detail. Existing precise-timing and incomplete-run tests remain, with nonzero module discovery in hosted CI. Source comparison, changed-call syntax formatting and staged whitespace pass locally. Root approved the design; source review and actual hosted execution remain pending. + +Root initial review confirmed the field/caller mapping and recorder-body preservation. The narrow follow-up removes six redundant field labels and the now-unused runner terminal-mutation import; the new duration assertion uses saturating subtraction. The production event builder remains unchanged and rejects a terminal timestamp preceding the start. Root narrow re-review accepted frozen tree a79c7c43eb66667ec8c0e97c5fc83e5dc9e93ab4. Hosted execution remains pending. From 48b728c5cc73d0227a3c596a7a8eab58308064f9 Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 07:47:14 -0700 Subject: [PATCH 22/25] refactor(workflow): group private session run context --- .github/workflows/quality-gates.yml | 5 ++ .../src/workflow/session_execution_api.rs | 48 ++++++----- .../src/workflow/session_scheduler_runner.rs | 84 ++++++++++--------- .../src/workflow/task_execution_owner.rs | 42 +++++++--- .../reports/2026-10-03-session-run-context.md | 7 ++ 5 files changed, 112 insertions(+), 74 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-session-run-context.md diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml index 93fcf95c6..a6074e552 100644 --- a/.github/workflows/quality-gates.yml +++ b/.github/workflows/quality-gates.yml @@ -251,6 +251,11 @@ jobs: - name: Run workflow-nodes tests run: cargo test -p workflow-nodes --lib + - name: Run complete session execution regression suite + run: | + cargo test -p pantograph-workflow-service --lib workflow::tests::session_execution:: -- --list > "$RUNNER_TEMP/session-execution-tests.list" + python3 -c 'import pathlib, sys; names = [line for line in pathlib.Path(sys.argv[1]).read_text().splitlines() if line.startswith("workflow::tests::session_execution::") and line.endswith(": test")]; print("Session execution tests discovered:", len(names)); sys.exit(0 if names else 1)' "$RUNNER_TEMP/session-execution-tests.list" + cargo test -p pantograph-workflow-service --lib workflow::tests::session_execution:: - name: Run terminal diagnostic regression suite run: | cargo test -p pantograph-workflow-service --lib workflow::runtime_branch_run_finalization::tests:: -- --list > "$RUNNER_TEMP/terminal-diagnostic-tests.list" diff --git a/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs b/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs index 5e0216ef5..8b6727ed0 100644 --- a/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs +++ b/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs @@ -48,8 +48,10 @@ use super::runtime_dispatch_assignment::{ WorkflowRuntimeDispatchAssignmentRepository, }; use super::session_io_artifacts::workflow_io_artifact_metadata; -use super::session_scheduler_runner::WorkflowSchedulerSessionRunner; -use super::task_execution_owner::WorkflowTaskExecutionOwner; +use super::session_scheduler_runner::{ + WorkflowSchedulerRunContext, WorkflowSchedulerSessionRunner, +}; +use super::task_execution_owner::{WorkflowNonRuntimeExecutionInput, WorkflowTaskExecutionOwner}; use super::task_execution_runtime::WorkflowTaskExecutionRuntimeOwner; use super::task_execution_worker::{ WorkflowTaskExecutionWorkerOutcome, WorkflowTaskExecutionWorkerRuntimeBranchCommand, @@ -415,12 +417,14 @@ impl WorkflowService { return WorkflowTaskExecutionOwner::run_non_runtime_to_completion( self, host, - &session, - run_snapshot.as_ref(), - &session_id, - &workflow_run_id, - &queued_run, - &scheduler_task_run_summary, + WorkflowNonRuntimeExecutionInput { + session: &session, + run_snapshot: run_snapshot.as_ref(), + session_id: &session_id, + workflow_run_id: &workflow_run_id, + queued_run: &queued_run, + summary: &scheduler_task_run_summary, + }, ) .await; } @@ -565,12 +569,14 @@ impl WorkflowService { Ok(Some(timeout_ms)) => { let run_future = runner.resume_runtime_dependency_readiness( host, - &session_id, - &workflow_run_id, - &active_run.workflow_id, - active_run.output_targets.as_deref(), - &scheduler_task_run_summary, - started_at, + WorkflowSchedulerRunContext { + session_id: &session_id, + workflow_run_id: &workflow_run_id, + workflow_id: &active_run.workflow_id, + output_targets: active_run.output_targets.as_deref(), + summary: &scheduler_task_run_summary, + started_at, + }, attempt_start_transition, ); match tokio::time::timeout(Duration::from_millis(timeout_ms), run_future).await { @@ -585,12 +591,14 @@ impl WorkflowService { runner .resume_runtime_dependency_readiness( host, - &session_id, - &workflow_run_id, - &active_run.workflow_id, - active_run.output_targets.as_deref(), - &scheduler_task_run_summary, - started_at, + WorkflowSchedulerRunContext { + session_id: &session_id, + workflow_run_id: &workflow_run_id, + workflow_id: &active_run.workflow_id, + output_targets: active_run.output_targets.as_deref(), + summary: &scheduler_task_run_summary, + started_at, + }, attempt_start_transition, ) .await diff --git a/crates/pantograph-workflow-service/src/workflow/session_scheduler_runner.rs b/crates/pantograph-workflow-service/src/workflow/session_scheduler_runner.rs index 31456e718..e89b18850 100644 --- a/crates/pantograph-workflow-service/src/workflow/session_scheduler_runner.rs +++ b/crates/pantograph-workflow-service/src/workflow/session_scheduler_runner.rs @@ -231,6 +231,16 @@ impl WorkflowPreDispatchPreparationOutcome { } } +#[derive(Debug, Clone, Copy)] +pub(super) struct WorkflowSchedulerRunContext<'a> { + pub(super) session_id: &'a str, + pub(super) workflow_run_id: &'a str, + pub(super) workflow_id: &'a str, + pub(super) output_targets: Option<&'a [WorkflowOutputTarget]>, + pub(super) summary: &'a WorkflowSchedulerTaskRunSummary, + pub(super) started_at: Instant, +} + struct ReadyRuntimeDispatchContext { task: WorkflowSchedulerTask, ready_record: SchedulerTaskStateRecord, @@ -244,14 +254,17 @@ impl<'a> WorkflowSchedulerSessionRunner<'a> { pub(super) async fn run_non_runtime_only( &self, host: &H, - session_id: &str, - workflow_run_id: &str, - workflow_id: &str, inputs: &[WorkflowPortBinding], - output_targets: Option<&[WorkflowOutputTarget]>, - summary: &WorkflowSchedulerTaskRunSummary, - started_at: Instant, + context: WorkflowSchedulerRunContext<'_>, ) -> Result { + let WorkflowSchedulerRunContext { + session_id, + workflow_run_id, + workflow_id, + output_targets, + summary, + started_at, + } = context; if !summary.is_non_runtime_only() || summary.has_runtime_inference() { return Err(WorkflowServiceError::Internal( "scheduler session runner received a runtime-containing run".to_string(), @@ -278,31 +291,22 @@ impl<'a> WorkflowSchedulerSessionRunner<'a> { pub(super) async fn resume_runtime_dependency_readiness( &self, host: &(impl WorkflowHost + ?Sized), - session_id: &str, - workflow_run_id: &str, - workflow_id: &str, - output_targets: Option<&[WorkflowOutputTarget]>, - summary: &WorkflowSchedulerTaskRunSummary, - started_at: Instant, + context: WorkflowSchedulerRunContext<'_>, attempt_start_transition: SchedulerTaskAttemptLifecycleTransition, ) -> Result { + let WorkflowSchedulerRunContext { + workflow_run_id, + summary, + .. + } = context; if !summary.has_runtime_inference() { return Err(WorkflowServiceError::InvalidRequest(format!( "workflow run '{}' is not a runtime inference run", workflow_run_id ))); } - self.continue_runtime_dependency_readiness( - host, - session_id, - workflow_run_id, - workflow_id, - output_targets, - summary, - started_at, - attempt_start_transition, - ) - .await + self.continue_runtime_dependency_readiness(host, context, attempt_start_transition) + .await } pub(super) async fn resume_progress_loop( @@ -318,14 +322,14 @@ impl<'a> WorkflowSchedulerSessionRunner<'a> { async fn continue_runtime_dependency_readiness( &self, host: &(impl WorkflowHost + ?Sized), - session_id: &str, - workflow_run_id: &str, - workflow_id: &str, - output_targets: Option<&[WorkflowOutputTarget]>, - summary: &WorkflowSchedulerTaskRunSummary, - started_at: Instant, + context: WorkflowSchedulerRunContext<'_>, attempt_start_transition: SchedulerTaskAttemptLifecycleTransition, ) -> Result { + let WorkflowSchedulerRunContext { + session_id, + workflow_run_id, + .. + } = context; let preparation = WorkflowPreDispatchPreparationBoundary::new(self.service) .prepare_runtime_dispatch(session_id, workflow_run_id) .await?; @@ -341,12 +345,7 @@ impl<'a> WorkflowSchedulerSessionRunner<'a> { } self.run_runtime_dispatch_ready_tasks( host, - session_id, - workflow_run_id, - workflow_id, - output_targets, - summary, - started_at, + context, preparation.admitted_runtime_readiness(), attempt_start_transition, ) @@ -804,15 +803,18 @@ impl<'a> WorkflowSchedulerSessionRunner<'a> { async fn run_runtime_dispatch_ready_tasks( &self, host: &(impl WorkflowHost + ?Sized), - session_id: &str, - workflow_run_id: &str, - workflow_id: &str, - output_targets: Option<&[WorkflowOutputTarget]>, - summary: &WorkflowSchedulerTaskRunSummary, - started_at: Instant, + context: WorkflowSchedulerRunContext<'_>, admitted_runtime_readiness: &[AdmittedRuntimeTaskReadiness], attempt_start_transition: SchedulerTaskAttemptLifecycleTransition, ) -> Result { + let WorkflowSchedulerRunContext { + session_id, + workflow_run_id, + workflow_id, + output_targets, + summary, + started_at, + } = context; let runtime_task_ids = runtime_task_ids_in_state(self.service, session_id, workflow_run_id, |kind| { kind == SchedulerTaskStateKind::Ready diff --git a/crates/pantograph-workflow-service/src/workflow/task_execution_owner.rs b/crates/pantograph-workflow-service/src/workflow/task_execution_owner.rs index e449be161..9f417c342 100644 --- a/crates/pantograph-workflow-service/src/workflow/task_execution_owner.rs +++ b/crates/pantograph-workflow-service/src/workflow/task_execution_owner.rs @@ -3,7 +3,9 @@ use std::time::Duration; use crate::scheduler::WorkflowExecutionSessionDequeuedRun; use pantograph_runtime_attribution::WorkflowRunSnapshotRecord; -use super::session_scheduler_runner::WorkflowSchedulerSessionRunner; +use super::session_scheduler_runner::{ + WorkflowSchedulerRunContext, WorkflowSchedulerSessionRunner, +}; use super::workflow_run_finalization::{ finalize_admitted_workflow_run, WorkflowRunFinalizationRequest, }; @@ -12,6 +14,15 @@ use super::{ WorkflowSchedulerTaskRunSummary, WorkflowService, WorkflowServiceError, }; +pub(super) struct WorkflowNonRuntimeExecutionInput<'a> { + pub(super) session: &'a WorkflowExecutionSessionSummary, + pub(super) run_snapshot: Option<&'a WorkflowRunSnapshotRecord>, + pub(super) session_id: &'a str, + pub(super) workflow_run_id: &'a str, + pub(super) queued_run: &'a WorkflowExecutionSessionDequeuedRun, + pub(super) summary: &'a WorkflowSchedulerTaskRunSummary, +} + pub(super) struct WorkflowTaskExecutionOwner; impl WorkflowTaskExecutionOwner { @@ -26,13 +37,16 @@ impl WorkflowTaskExecutionOwner { pub(super) async fn run_non_runtime_to_completion( service: &WorkflowService, host: &H, - session: &WorkflowExecutionSessionSummary, - run_snapshot: Option<&WorkflowRunSnapshotRecord>, - session_id: &str, - workflow_run_id: &str, - queued_run: &WorkflowExecutionSessionDequeuedRun, - summary: &WorkflowSchedulerTaskRunSummary, + input: WorkflowNonRuntimeExecutionInput<'_>, ) -> Result { + let WorkflowNonRuntimeExecutionInput { + session, + run_snapshot, + session_id, + workflow_run_id, + queued_run, + summary, + } = input; service.record_run_started_event_if_configured(session, run_snapshot, queued_run)?; let run_started_at = std::time::Instant::now(); let queued_workflow_semantic_version = queued_run.queued.workflow_semantic_version.clone(); @@ -40,13 +54,15 @@ impl WorkflowTaskExecutionOwner { let runner = WorkflowSchedulerSessionRunner::new(service); let run_future = runner.run_non_runtime_only( host, - session_id, - workflow_run_id, - &queued_run.workflow_id, &queued_run.queued.inputs, - queued_run.queued.output_targets.as_deref(), - summary, - run_started_at, + WorkflowSchedulerRunContext { + session_id, + workflow_run_id, + workflow_id: &queued_run.workflow_id, + output_targets: queued_run.queued.output_targets.as_deref(), + summary, + started_at: run_started_at, + }, ); let run_result = if let Some(timeout_ms) = queued_run.queued.timeout_ms { match tokio::time::timeout(Duration::from_millis(timeout_ms), run_future).await { diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-session-run-context.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-session-run-context.md new file mode 100644 index 000000000..dd5d1cd58 --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-session-run-context.md @@ -0,0 +1,7 @@ +# Private session runner and non-runtime execution inputs + +Group the six values shared by four high-arity runner methods into private WorkflowSchedulerRunContext: borrowed session_id, workflow_run_id, workflow_id, output_targets and summary, plus the original Instant. Keep host, inputs, admitted readiness and attempt transition separate. There are five runner calls: one non-runtime entry, two resume entries and two internal forwards. The Copy context contains only existing borrowed values and Instant; no allocation, new clock read or cloned run data is introduced. + +The sole WorkflowTaskExecutionOwner caller receives a separate six-field WorkflowNonRuntimeExecutionInput: session, run_snapshot, session_id, workflow_run_id, queued_run and summary. Existing finalization requests carry extra result/policy values and are not reused. All existing service/host parameters, borrowing and result signatures remain. Timeout, cancellation cleanup, finalization, no-load behavior and runtime readiness decisions are unchanged. + +Programmatic source comparisons confirm unchanged non-runtime execution/finalization, ready-dispatch body and owner timeout/finalization block. CI explicitly discovers and runs the entire session-execution module, including non-runtime lifecycle, timeout, readiness defer/resume, bootstrap recovery and no-runtime-load cases, with a nonzero guard. Root approved this bounded design. Whitespace/changed-call syntax checks pass locally; root source review accepted all five files at frozen tree 66a25e87fe897ea1001622c0768a51cc51d07882; hosted execution remains pending. From 533a80b9650cff2c1b4f9a8b9087760139fe89c9 Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 07:52:07 -0700 Subject: [PATCH 23/25] refactor(workflow): clarify private projection and summary types --- .github/workflows/quality-gates.yml | 5 ++ .../src/workflow/task_graph.rs | 16 ++++--- .../src/workflow/task_run_summary.rs | 46 +++++++++++++++---- .../2026-10-03-workflow-private-type-names.md | 7 +++ 4 files changed, 59 insertions(+), 15 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-workflow-private-type-names.md diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml index a6074e552..52fb41d4b 100644 --- a/.github/workflows/quality-gates.yml +++ b/.github/workflows/quality-gates.yml @@ -251,6 +251,11 @@ jobs: - name: Run workflow-nodes tests run: cargo test -p workflow-nodes --lib + - name: Run task summary contract tests + run: | + cargo test -p pantograph-workflow-service --lib workflow::task_run_summary::tests:: -- --list > "$RUNNER_TEMP/task-summary-tests.list" + python3 -c 'import pathlib, sys; names = [line for line in pathlib.Path(sys.argv[1]).read_text().splitlines() if line.startswith("workflow::task_run_summary::tests::") and line.endswith(": test")]; print("Task summary tests discovered:", len(names)); sys.exit(0 if names else 1)' "$RUNNER_TEMP/task-summary-tests.list" + cargo test -p pantograph-workflow-service --lib workflow::task_run_summary::tests:: - name: Run complete session execution regression suite run: | cargo test -p pantograph-workflow-service --lib workflow::tests::session_execution:: -- --list > "$RUNNER_TEMP/session-execution-tests.list" diff --git a/crates/pantograph-workflow-service/src/workflow/task_graph.rs b/crates/pantograph-workflow-service/src/workflow/task_graph.rs index 89b9e71f5..267a1169e 100644 --- a/crates/pantograph-workflow-service/src/workflow/task_graph.rs +++ b/crates/pantograph-workflow-service/src/workflow/task_graph.rs @@ -259,6 +259,14 @@ fn dependency_task_ids(bindings: &[WorkflowSchedulerTaskInputBinding]) -> Vec, + Option, + Option, + Option, + Vec, +); + fn schedulable_intent_for_node( workflow_id: &SchedulerWorkflowId, workflow_run_id: &SchedulerWorkflowRunId, @@ -266,13 +274,7 @@ fn schedulable_intent_for_node( task_id: &SchedulerTaskId, execution_class: WorkflowSchedulerTaskExecutionClass, inference_task_projection: Option<&WorkflowSchedulerInferenceTaskProjection>, -) -> ( - Option, - Option, - Option, - Option, - Vec, -) { +) -> SchedulableIntentProjection { if execution_class != WorkflowSchedulerTaskExecutionClass::RuntimeInference { return (None, None, None, None, Vec::new()); } diff --git a/crates/pantograph-workflow-service/src/workflow/task_run_summary.rs b/crates/pantograph-workflow-service/src/workflow/task_run_summary.rs index cbad3add0..e085ba2ae 100644 --- a/crates/pantograph-workflow-service/src/workflow/task_run_summary.rs +++ b/crates/pantograph-workflow-service/src/workflow/task_run_summary.rs @@ -54,7 +54,7 @@ pub(crate) fn workflow_scheduler_task_run_summary( .iter() .find(|record| record.task_id.as_str() == task_id) else { - return Err(WorkflowSchedulerTaskRunSummaryError::MissingTaskState { + return Err(WorkflowSchedulerTaskRunSummaryError::Missing { task_id: task_id.to_string(), }); }; @@ -62,7 +62,7 @@ pub(crate) fn workflow_scheduler_task_run_summary( || record.workflow_run_id.as_str() != task.workflow_run_id.as_str() || record.node_id.as_str() != task.node_id.as_str() { - return Err(WorkflowSchedulerTaskRunSummaryError::MismatchedTaskState { + return Err(WorkflowSchedulerTaskRunSummaryError::Mismatched { task_id: task_id.to_string(), }); } @@ -95,7 +95,7 @@ pub(crate) fn workflow_scheduler_task_run_summary( } if let Some(extra_task_id) = record_task_ids.into_iter().next() { - return Err(WorkflowSchedulerTaskRunSummaryError::UnexpectedTaskState { + return Err(WorkflowSchedulerTaskRunSummaryError::Unexpected { task_id: extra_task_id.to_string(), }); } @@ -107,11 +107,11 @@ pub(crate) fn workflow_scheduler_task_run_summary( #[non_exhaustive] pub(crate) enum WorkflowSchedulerTaskRunSummaryError { #[error("scheduler task '{task_id}' has no active task-state record")] - MissingTaskState { task_id: String }, + Missing { task_id: String }, #[error("active task-state record for scheduler task '{task_id}' has mismatched correlation")] - MismatchedTaskState { task_id: String }, + Mismatched { task_id: String }, #[error("active task-state record exists for unknown scheduler task '{task_id}'")] - UnexpectedTaskState { task_id: String }, + Unexpected { task_id: String }, } #[cfg(test)] @@ -269,10 +269,36 @@ mod tests { )]); let error = workflow_scheduler_task_run_summary(&graph, &[]).expect_err("missing record"); + assert_eq!( + error.to_string(), + "scheduler task 'prompt' has no active task-state record" + ); + + assert_eq!( + error, + WorkflowSchedulerTaskRunSummaryError::Missing { + task_id: "prompt".to_string() + } + ); + } + #[test] + fn rejects_mismatched_task_state_with_unchanged_display() { + let graph = task_graph(&[( + "prompt", + WorkflowSchedulerTaskExecutionClass::NonRuntimeNodeEngine, + )]); + let mut mismatched = record("prompt", awaiting_inputs()); + mismatched.workflow_run_id = SchedulerWorkflowRunId::parse("other-run").unwrap(); + let error = workflow_scheduler_task_run_summary(&graph, &[mismatched]) + .expect_err("mismatched correlation"); + assert_eq!( + error.to_string(), + "active task-state record for scheduler task 'prompt' has mismatched correlation" + ); assert_eq!( error, - WorkflowSchedulerTaskRunSummaryError::MissingTaskState { + WorkflowSchedulerTaskRunSummaryError::Mismatched { task_id: "prompt".to_string() } ); @@ -293,10 +319,14 @@ mod tests { ], ) .expect_err("unexpected record"); + assert_eq!( + error.to_string(), + "active task-state record exists for unknown scheduler task 'extra'" + ); assert_eq!( error, - WorkflowSchedulerTaskRunSummaryError::UnexpectedTaskState { + WorkflowSchedulerTaskRunSummaryError::Unexpected { task_id: "extra".to_string() } ); diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-workflow-private-type-names.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-workflow-private-type-names.md new file mode 100644 index 000000000..292c07e47 --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-workflow-private-type-names.md @@ -0,0 +1,7 @@ +# Private projection tuple and summary error names + +Name the existing five-element schedulable projection tuple with the private SchedulableIntentProjection alias; tuple members, order, function body and caller destructuring are unchanged. Shorten only the crate-private summary error variants from MissingTaskState/MismatchedTaskState/UnexpectedTaskState to Missing/Mismatched/Unexpected, retaining every task_id field and exact thiserror Display string. Neither type has a new public or serialized representation. + +The source inventory finds two production summary consumers, both using Display interpolation before admission/resume; all variant matches are local tests. Derived Debug variant labels intentionally change with the private names and may appear in test failures, but no production Debug consumer was found. Real missing/unexpected summary tests now assert exact Display, and a real mismatched-run correlation test covers the third branch. CI runs the complete summary module with nonzero discovery, retaining projection tests for the tuple alias. + +Root approved this bounded cleanup. Root source review accepted all four files at frozen tree c3df7026f434651b5b57876d967110d55f3150c6. Hosted execution remains pending. No local Rust execution is claimed. From c369cbe4b12b563eb8a5d4419e6e0ed729721137 Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 08:06:30 -0700 Subject: [PATCH 24/25] refactor(registry): box private selection diagnostic errors --- .github/workflows/quality-gates.yml | 2 + .../src/runtime_selection_policy.rs | 51 +++++++++++++++---- .../src/technical_fit_tests.rs | 9 ++++ ...0-03-runtime-registry-diagnostic-errors.md | 7 +++ 4 files changed, 58 insertions(+), 11 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-runtime-registry-diagnostic-errors.md diff --git a/.github/workflows/quality-gates.yml b/.github/workflows/quality-gates.yml index 52fb41d4b..2edc2cb9d 100644 --- a/.github/workflows/quality-gates.yml +++ b/.github/workflows/quality-gates.yml @@ -251,6 +251,8 @@ jobs: - name: Run workflow-nodes tests run: cargo test -p workflow-nodes --lib + - name: Run runtime registry contract tests + run: cargo test -p pantograph-runtime-registry - name: Run task summary contract tests run: | cargo test -p pantograph-workflow-service --lib workflow::task_run_summary::tests:: -- --list > "$RUNNER_TEMP/task-summary-tests.list" diff --git a/crates/pantograph-runtime-registry/src/runtime_selection_policy.rs b/crates/pantograph-runtime-registry/src/runtime_selection_policy.rs index 2eb78250f..a696b5284 100644 --- a/crates/pantograph-runtime-registry/src/runtime_selection_policy.rs +++ b/crates/pantograph-runtime-registry/src/runtime_selection_policy.rs @@ -178,7 +178,7 @@ pub(crate) fn select_runtime_technical_fit_automatically( return RuntimeSelectionDecision::new(unselected_decision_with_device_diagnostics( RuntimeTechnicalFitSelectionMode::Automatic, Vec::new(), - vec![diagnostic], + vec![*diagnostic], )); } }; @@ -292,7 +292,7 @@ fn automatic_selection_policy_trace( controlled_exploration_seed_basis: Option<&str>, history_threshold_state: RuntimeTechnicalFitHistoryThresholdState, history_ranking_enabled: bool, -) -> Result { +) -> Result> { let candidate_set_summary = automatic_candidate_set_summary(request, eligible_candidates)?; Ok(RuntimeTechnicalFitSelectionPolicyTrace { policy_version: TECHNICAL_FIT_SELECTION_POLICY_VERSION, @@ -327,12 +327,12 @@ fn automatic_selection_policy_trace( fn automatic_candidate_set_summary( request: &RuntimeTechnicalFitRequest, eligible_candidates: &[&RuntimeTechnicalFitCandidate], -) -> Result { +) -> Result> { let total_candidate_count = checked_candidate_count(request.candidates.len())?; let eligible_candidate_count = checked_candidate_count(eligible_candidates.len())?; let rejected_candidate_count = total_candidate_count .checked_sub(eligible_candidate_count) - .ok_or_else(candidate_summary_count_diagnostic)?; + .ok_or_else(|| Box::new(candidate_summary_count_diagnostic()))?; Ok(RuntimeTechnicalFitCandidateSetSummary { total_candidate_count, @@ -346,8 +346,8 @@ fn automatic_candidate_set_summary( .normalized()) } -fn checked_candidate_count(count: usize) -> Result { - u32::try_from(count).map_err(|_| candidate_summary_count_diagnostic()) +fn checked_candidate_count(count: usize) -> Result> { + u32::try_from(count).map_err(|_| Box::new(candidate_summary_count_diagnostic())) } fn candidate_summary_count_diagnostic() -> RuntimeTechnicalFitDeviceDiagnostic { @@ -822,7 +822,7 @@ fn resource_budget_diagnostic( }; let reserved_bytes = match active_reserved_bytes(runtime_snapshot, resource_kind) { Ok(reserved_bytes) => reserved_bytes, - Err(diagnostic) => return Some(diagnostic_context_from_candidate(diagnostic, candidate)), + Err(diagnostic) => return Some(diagnostic_context_from_candidate(*diagnostic, candidate)), }; let Some(available_bytes) = after_safety_margin_bytes.checked_sub(reserved_bytes) else { return Some(resource_budget_underflow_diagnostic( @@ -871,7 +871,7 @@ fn resource_budget_diagnostic( fn active_reserved_bytes( runtime_snapshot: &RuntimeRegistryRuntimeSnapshot, resource_kind: RuntimeAdmissionResourceKind, -) -> Result { +) -> Result> { let mut reserved_bytes = 0_u64; for active_claim in &runtime_snapshot.active_reservation_claims { for claim in active_claim @@ -891,10 +891,10 @@ fn checked_add_claim_bytes( resource_kind: RuntimeAdmissionResourceKind, reserved_bytes: u64, claim: &RuntimeReservationResourceClaim, -) -> Result { +) -> Result> { reserved_bytes .checked_add(claim.bytes) - .ok_or_else(|| RuntimeTechnicalFitDeviceDiagnostic { + .ok_or_else(|| Box::new(RuntimeTechnicalFitDeviceDiagnostic { code: RuntimeTechnicalFitDeviceDiagnosticCode::ResourceAccountingOverflow, severity: RuntimeTechnicalFitDeviceDiagnosticSeverity::Error, message: format!( @@ -911,7 +911,7 @@ fn checked_add_claim_bytes( model_id: None, evidence_key: Some(resource_kind.resource_label().to_string()), requested_runtime_key: None, - }) + })) } fn resource_budget_underflow_diagnostic( @@ -1132,3 +1132,32 @@ fn candidate_has_available_peak_memory_estimate(candidate: &RuntimeTechnicalFitC && estimate.value_bytes().is_some() }) } + +#[cfg(test)] +mod private_diagnostic_tests { + use super::checked_candidate_count; + + #[test] + fn candidate_counts_preserve_exact_valid_boundaries() { + assert_eq!(checked_candidate_count(0).unwrap(), 0); + assert_eq!( + checked_candidate_count(u32::MAX as usize).unwrap(), + u32::MAX + ); + } + + #[test] + #[cfg(target_pointer_width = "64")] + fn candidate_count_overflow_preserves_exact_diagnostic_json() { + let diagnostic = checked_candidate_count(u32::MAX as usize + 1).unwrap_err(); + let expected = serde_json::json!({ + "code": "no_valid_candidate", + "severity": "error", + "message": "technical-fit candidate set is too large to summarize exactly" + }); + assert_eq!(serde_json::to_value(diagnostic.as_ref()).unwrap(), expected); + let decoded: crate::technical_fit::RuntimeTechnicalFitDeviceDiagnostic = + serde_json::from_value(expected).unwrap(); + assert_eq!(*diagnostic, decoded); + } +} diff --git a/crates/pantograph-runtime-registry/src/technical_fit_tests.rs b/crates/pantograph-runtime-registry/src/technical_fit_tests.rs index 33fb0bc60..ad0971b37 100644 --- a/crates/pantograph-runtime-registry/src/technical_fit_tests.rs +++ b/crates/pantograph-runtime-registry/src/technical_fit_tests.rs @@ -2563,6 +2563,15 @@ fn selector_reports_resource_accounting_overflow_before_selection() { assert_eq!(decision.selected_candidate_id, None); assert_eq!(decision.device_diagnostics.len(), 1); + assert_eq!(serde_json::to_value(&decision.device_diagnostics[0]).unwrap(), serde_json::json!({ + "code": "resource_accounting_overflow", + "severity": "error", + "message": "technical-fit resource accounting overflowed while summing active ram_bytes claims for runtime 'runtime-overflow'", + "runtime_id": "runtime-overflow", + "backend_key": "pytorch", + "evidence_key": "ram_bytes", + "requested_runtime_key": "pytorch" + })); assert_eq!( decision.device_diagnostics[0].code, RuntimeTechnicalFitDeviceDiagnosticCode::ResourceAccountingOverflow diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-runtime-registry-diagnostic-errors.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-runtime-registry-diagnostic-errors.md new file mode 100644 index 000000000..345a7a049 --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-runtime-registry-diagnostic-errors.md @@ -0,0 +1,7 @@ +# Private runtime selection diagnostic errors + +Fresh PR #39 aggregate Clippy clears workflow-service and stops at five private runtime-registry Result paths whose error is the same 224-byte RuntimeTechnicalFitDeviceDiagnostic DTO. Box only these errors in automatic_selection_policy_trace, automatic_candidate_set_summary, checked_candidate_count, active_reserved_bytes and checked_add_claim_bytes. Unwrap at the existing automatic decision-vector and resource-budget context adapters. Preserve public signatures, the raw DTO, diagnostic fields/JSON, ranking/accounting decisions and success values. + +Boundary tests retain zero/u32-max candidate counts and exact overflow diagnostic JSON (the usize overflow case is explicitly a 64-bit test). The existing public selector resource-overflow test now checks the entire contextualized payload, while full registry tests retain successful selection, budget and underflow behavior. Only failing paths add a box allocation. No broad Result or public layout change is made. + +Root approved this bounded registry branch using the existing checkout. Root source review accepted all four files at frozen tree e88e208b5567da1dda64974dd1c65bb4524d5801. Hosted full-crate/aggregate qualification remains pending; no local Rust execution is claimed. From 2a4d5c8735188e18e2ab2e085b73c48bf7362807 Mon Sep 17 00:00:00 2001 From: Puma Date: Sat, 3 Oct 2026 08:16:19 -0700 Subject: [PATCH 25/25] style(rust): apply pinned formatter to repair milestones --- .../src/tests.rs | 8 +-- .../src/node_execution_ledger.rs | 70 ++++++++++--------- .../src/technical_fit_tests.rs | 21 +++--- .../src/graph/inference_validation_state.rs | 8 +-- .../src/scheduler/task_orchestrator.rs | 6 +- .../executable_validation_snapshot.rs | 8 +-- .../src/workflow/session_execution_api.rs | 8 +-- .../src/workflow/session_scheduler_runner.rs | 44 ++++++------ .../src/workflow/task_execution_worker.rs | 22 +++--- .../workflow/tests/task_binding_resolution.rs | 8 +-- .../src/workflow/tests/task_graph.rs | 8 +-- .../2026-10-03-pinned-rustfmt-correction.md | 21 ++++++ 12 files changed, 132 insertions(+), 100 deletions(-) create mode 100644 docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-pinned-rustfmt-correction.md diff --git a/crates/pantograph-diagnostics-ledger/src/tests.rs b/crates/pantograph-diagnostics-ledger/src/tests.rs index 2948baa34..5d8ff19ba 100644 --- a/crates/pantograph-diagnostics-ledger/src/tests.rs +++ b/crates/pantograph-diagnostics-ledger/src/tests.rs @@ -6485,8 +6485,8 @@ fn sample_inference_execution_diagnostic_event() -> DiagnosticEventAppendRequest privacy_class: DiagnosticEventPrivacyClass::SystemMetadata, retention_class: DiagnosticEventRetentionClass::AuditMetadata, payload_ref: None, - payload: DiagnosticEventPayload::InferenceExecutionDiagnosticObserved( - Box::new(InferenceExecutionDiagnosticObservedPayload { + payload: DiagnosticEventPayload::InferenceExecutionDiagnosticObserved(Box::new( + InferenceExecutionDiagnosticObservedPayload { request_id: "req-a".to_string(), task_id: "text_generation".to_string(), lifecycle_phase: Some("task_validation".to_string()), @@ -6552,8 +6552,8 @@ fn sample_inference_execution_diagnostic_event() -> DiagnosticEventAppendRequest message: Some("not mapped by this backend boundary".to_string()), }, ], - }), - ), + }, + )), } } diff --git a/crates/pantograph-embedded-runtime/src/node_execution_ledger.rs b/crates/pantograph-embedded-runtime/src/node_execution_ledger.rs index 183fa5f49..9abde039f 100644 --- a/crates/pantograph-embedded-runtime/src/node_execution_ledger.rs +++ b/crates/pantograph-embedded-runtime/src/node_execution_ledger.rs @@ -532,28 +532,30 @@ impl NodeExecutionWorkflowLedgerSink { privacy_class: artifact.privacy_class, retention_class: artifact.retention_class, payload_ref: artifact.payload_ref, - payload: DiagnosticEventPayload::IoArtifactObserved(Box::new(IoArtifactObservedPayload { - artifact_fact_id: Some(artifact.artifact_fact_id), - payload_artifact_id: Some(artifact.payload_artifact_id), - artifact_id: artifact.artifact_id, - artifact_role, - logical_payload_lineage_id: Some(artifact.logical_payload_lineage_id), - producer_node_id, - producer_port_id, - consumer_node_id, - consumer_port_id, - media_type: artifact.media_type, - size_bytes: artifact.size_bytes, - content_hash: artifact.content_hash, - retention_state: Some(artifact.retention_state), - retention_reason: artifact.retention_reason, - payload_kind: artifact.payload_kind, - lifecycle_state: artifact.lifecycle_state, - access_modes: artifact.access_modes, - read_handle: artifact.read_handle, - stream_handle: None, - format: artifact.format, - })), + payload: DiagnosticEventPayload::IoArtifactObserved(Box::new( + IoArtifactObservedPayload { + artifact_fact_id: Some(artifact.artifact_fact_id), + payload_artifact_id: Some(artifact.payload_artifact_id), + artifact_id: artifact.artifact_id, + artifact_role, + logical_payload_lineage_id: Some(artifact.logical_payload_lineage_id), + producer_node_id, + producer_port_id, + consumer_node_id, + consumer_port_id, + media_type: artifact.media_type, + size_bytes: artifact.size_bytes, + content_hash: artifact.content_hash, + retention_state: Some(artifact.retention_state), + retention_reason: artifact.retention_reason, + payload_kind: artifact.payload_kind, + lifecycle_state: artifact.lifecycle_state, + access_modes: artifact.access_modes, + read_handle: artifact.read_handle, + stream_handle: None, + format: artifact.format, + }, + )), }; self.workflow_service @@ -951,8 +953,8 @@ fn build_kv_cache_diagnostic_event_ledger_append_request( privacy_class: DiagnosticEventPrivacyClass::SystemMetadata, retention_class: DiagnosticEventRetentionClass::AuditMetadata, payload_ref: None, - payload: DiagnosticEventPayload::InferenceExecutionDiagnosticObserved( - Box::new(InferenceExecutionDiagnosticObservedPayload { + payload: DiagnosticEventPayload::InferenceExecutionDiagnosticObserved(Box::new( + InferenceExecutionDiagnosticObservedPayload { request_id: format!("{task_id}:kv_cache"), task_id: "kv_cache".to_string(), lifecycle_phase: Some("kv_cache".to_string()), @@ -981,8 +983,8 @@ fn build_kv_cache_diagnostic_event_ledger_append_request( .take(MAX_INFERENCE_OPTION_DIAGNOSTICS) .map(kv_cache_option_diagnostic_summary) .collect(), - }), - ), + }, + )), }) } @@ -1047,8 +1049,8 @@ fn build_runtime_settings_diagnostic_event_ledger_append_request( privacy_class: DiagnosticEventPrivacyClass::SystemMetadata, retention_class: DiagnosticEventRetentionClass::AuditMetadata, payload_ref: None, - payload: DiagnosticEventPayload::InferenceExecutionDiagnosticObserved( - Box::new(InferenceExecutionDiagnosticObservedPayload { + payload: DiagnosticEventPayload::InferenceExecutionDiagnosticObserved(Box::new( + InferenceExecutionDiagnosticObservedPayload { request_id: format!("{task_id}:runtime_settings"), task_id: "runtime_settings".to_string(), lifecycle_phase: Some("runtime_settings".to_string()), @@ -1075,8 +1077,8 @@ fn build_runtime_settings_diagnostic_event_ledger_append_request( compatibility_issues: Vec::new(), option_support_counts: InferenceOptionSupportCounts::default(), option_diagnostics: Vec::new(), - }), - ), + }, + )), }) } @@ -1146,8 +1148,8 @@ fn build_inference_diagnostic_event_ledger_append_request( privacy_class: DiagnosticEventPrivacyClass::SystemMetadata, retention_class: DiagnosticEventRetentionClass::AuditMetadata, payload_ref: None, - payload: DiagnosticEventPayload::InferenceExecutionDiagnosticObserved( - Box::new(InferenceExecutionDiagnosticObservedPayload { + payload: DiagnosticEventPayload::InferenceExecutionDiagnosticObserved(Box::new( + InferenceExecutionDiagnosticObservedPayload { request_id: event .request_id .clone() @@ -1196,8 +1198,8 @@ fn build_inference_diagnostic_event_ledger_append_request( .take(MAX_INFERENCE_OPTION_DIAGNOSTICS) .map(option_diagnostic_summary) .collect(), - }), - ), + }, + )), }) } diff --git a/crates/pantograph-runtime-registry/src/technical_fit_tests.rs b/crates/pantograph-runtime-registry/src/technical_fit_tests.rs index ad0971b37..570112317 100644 --- a/crates/pantograph-runtime-registry/src/technical_fit_tests.rs +++ b/crates/pantograph-runtime-registry/src/technical_fit_tests.rs @@ -2563,15 +2563,18 @@ fn selector_reports_resource_accounting_overflow_before_selection() { assert_eq!(decision.selected_candidate_id, None); assert_eq!(decision.device_diagnostics.len(), 1); - assert_eq!(serde_json::to_value(&decision.device_diagnostics[0]).unwrap(), serde_json::json!({ - "code": "resource_accounting_overflow", - "severity": "error", - "message": "technical-fit resource accounting overflowed while summing active ram_bytes claims for runtime 'runtime-overflow'", - "runtime_id": "runtime-overflow", - "backend_key": "pytorch", - "evidence_key": "ram_bytes", - "requested_runtime_key": "pytorch" - })); + assert_eq!( + serde_json::to_value(&decision.device_diagnostics[0]).unwrap(), + serde_json::json!({ + "code": "resource_accounting_overflow", + "severity": "error", + "message": "technical-fit resource accounting overflowed while summing active ram_bytes claims for runtime 'runtime-overflow'", + "runtime_id": "runtime-overflow", + "backend_key": "pytorch", + "evidence_key": "ram_bytes", + "requested_runtime_key": "pytorch" + }) + ); assert_eq!( decision.device_diagnostics[0].code, RuntimeTechnicalFitDeviceDiagnosticCode::ResourceAccountingOverflow diff --git a/crates/pantograph-workflow-service/src/graph/inference_validation_state.rs b/crates/pantograph-workflow-service/src/graph/inference_validation_state.rs index 000e8c783..5b5186a60 100644 --- a/crates/pantograph-workflow-service/src/graph/inference_validation_state.rs +++ b/crates/pantograph-workflow-service/src/graph/inference_validation_state.rs @@ -1467,8 +1467,8 @@ impl CurrentInferenceValidationNodeRecord { message: format!("dependency requirements proof is not current: {error:?}"), }, )?; - return Ok(WorkflowSchedulerInferenceTaskProjection::Ready( - Box::new(WorkflowSchedulerReadyInferenceTaskProjection { + return Ok(WorkflowSchedulerInferenceTaskProjection::Ready(Box::new( + WorkflowSchedulerReadyInferenceTaskProjection { node_id: pantograph_scheduler::SchedulerNodeId::parse(self.node_id.as_str()) .map_err(|error| { CurrentInferenceSchedulerProjectionError::IncompleteNodeState { @@ -1532,8 +1532,8 @@ impl CurrentInferenceValidationNodeRecord { .dependency_override_fingerprint .clone(), }, - }), - )); + }, + ))); } Ok(WorkflowSchedulerInferenceTaskProjection::Blocked( diff --git a/crates/pantograph-workflow-service/src/scheduler/task_orchestrator.rs b/crates/pantograph-workflow-service/src/scheduler/task_orchestrator.rs index 47365ce4b..34c3167df 100644 --- a/crates/pantograph-workflow-service/src/scheduler/task_orchestrator.rs +++ b/crates/pantograph-workflow-service/src/scheduler/task_orchestrator.rs @@ -2115,9 +2115,9 @@ fn dispatch_selected_handoff_from_selection( ) -> Result { if selection.state != SchedulerDispatchSelectionState::Selected { return Err( - WorkflowSchedulerTaskOrchestratorError::RuntimeDispatchSelectionNoSelection( - Box::new(selection), - ), + WorkflowSchedulerTaskOrchestratorError::RuntimeDispatchSelectionNoSelection(Box::new( + selection, + )), ); } let Some(dispatch_decision) = selection.dispatch_decision else { diff --git a/crates/pantograph-workflow-service/src/workflow/executable_validation_snapshot.rs b/crates/pantograph-workflow-service/src/workflow/executable_validation_snapshot.rs index 7c35fa013..ea29b7d88 100644 --- a/crates/pantograph-workflow-service/src/workflow/executable_validation_snapshot.rs +++ b/crates/pantograph-workflow-service/src/workflow/executable_validation_snapshot.rs @@ -491,8 +491,8 @@ impl WorkflowExecutableValidationSnapshotNode { } })?; - Ok(WorkflowSchedulerInferenceTaskProjection::Ready( - Box::new(WorkflowSchedulerReadyInferenceTaskProjection { + Ok(WorkflowSchedulerInferenceTaskProjection::Ready(Box::new( + WorkflowSchedulerReadyInferenceTaskProjection { node_id: scheduler_node_id, descriptor_fingerprint: self.descriptor_fingerprint.clone(), task_type, @@ -504,8 +504,8 @@ impl WorkflowExecutableValidationSnapshotNode { dependency_readiness_source: workflow_scheduler_dependency_readiness_source( snapshot, self, )?, - }), - )) + }, + ))) } } diff --git a/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs b/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs index 8b6727ed0..921c7ef0c 100644 --- a/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs +++ b/crates/pantograph-workflow-service/src/workflow/session_execution_api.rs @@ -1353,8 +1353,8 @@ impl WorkflowService { privacy_class: metadata.privacy_class, retention_class: metadata.retention_class, payload_ref: metadata.payload_ref.clone(), - payload: DiagnosticEventPayload::IoArtifactObserved( - Box::new(IoArtifactObservedPayload { + payload: DiagnosticEventPayload::IoArtifactObserved(Box::new( + IoArtifactObservedPayload { artifact_fact_id: Some(metadata.artifact_fact_id), payload_artifact_id: Some(metadata.payload_artifact_id), artifact_id: metadata.artifact_id, @@ -1391,8 +1391,8 @@ impl WorkflowService { read_handle: metadata.read_handle, stream_handle: metadata.stream_handle, format: metadata.format, - }), - ), + }, + )), }, )?; } diff --git a/crates/pantograph-workflow-service/src/workflow/session_scheduler_runner.rs b/crates/pantograph-workflow-service/src/workflow/session_scheduler_runner.rs index e89b18850..bf65931b2 100644 --- a/crates/pantograph-workflow-service/src/workflow/session_scheduler_runner.rs +++ b/crates/pantograph-workflow-service/src/workflow/session_scheduler_runner.rs @@ -513,16 +513,18 @@ impl<'a> WorkflowSchedulerSessionRunner<'a> { "scheduler non-runtime task completion failed: {error}" )) })?; - self.record_scheduler_task_attempt_terminal(WorkflowSchedulerTaskAttemptTerminalInput { - task: started.task(), - attempt_id: started.attempt_id().as_str(), - started_at_ms: started.started_at_ms(), - transition: SchedulerTaskAttemptLifecycleTransition::Completed, - reason: "scheduler task attempt completed", - error_summary: None, - selected_dispatch: None, - terminal_mutation: None, - })?; + self.record_scheduler_task_attempt_terminal( + WorkflowSchedulerTaskAttemptTerminalInput { + task: started.task(), + attempt_id: started.attempt_id().as_str(), + started_at_ms: started.started_at_ms(), + transition: SchedulerTaskAttemptLifecycleTransition::Completed, + reason: "scheduler task attempt completed", + error_summary: None, + selected_dispatch: None, + terminal_mutation: None, + }, + )?; } Err( crate::scheduler::WorkflowSchedulerTaskOrchestratorError::NonRuntimeTaskAdapter( @@ -541,16 +543,18 @@ impl<'a> WorkflowSchedulerSessionRunner<'a> { &error, ); if failed.is_ok() { - self.record_scheduler_task_attempt_terminal(WorkflowSchedulerTaskAttemptTerminalInput { - task: started.task(), - attempt_id: started.attempt_id().as_str(), - started_at_ms: started.started_at_ms(), - transition: SchedulerTaskAttemptLifecycleTransition::Failed, - reason: "scheduler non-runtime task execution failed", - error_summary: Some(error.to_string()), - selected_dispatch: None, - terminal_mutation: None, - })?; + self.record_scheduler_task_attempt_terminal( + WorkflowSchedulerTaskAttemptTerminalInput { + task: started.task(), + attempt_id: started.attempt_id().as_str(), + started_at_ms: started.started_at_ms(), + transition: SchedulerTaskAttemptLifecycleTransition::Failed, + reason: "scheduler non-runtime task execution failed", + error_summary: Some(error.to_string()), + selected_dispatch: None, + terminal_mutation: None, + }, + )?; } return Err(WorkflowServiceError::InvalidRequest(format!( "scheduler non-runtime task execution failed: {error}" diff --git a/crates/pantograph-workflow-service/src/workflow/task_execution_worker.rs b/crates/pantograph-workflow-service/src/workflow/task_execution_worker.rs index e632a73db..2f2116309 100644 --- a/crates/pantograph-workflow-service/src/workflow/task_execution_worker.rs +++ b/crates/pantograph-workflow-service/src/workflow/task_execution_worker.rs @@ -1124,12 +1124,14 @@ impl WorkflowTaskExecutionWorkerOutcome { response: WorkflowRunResponse, diagnostics: Vec, ) -> Self { - Self::RuntimeBranchCompleted(Box::new(WorkflowTaskExecutionWorkerRuntimeBranchCompletedOutcome { - session_id: command.session_id.clone(), - workflow_run_id: command.workflow_run_id.clone(), - response, - diagnostics, - })) + Self::RuntimeBranchCompleted(Box::new( + WorkflowTaskExecutionWorkerRuntimeBranchCompletedOutcome { + session_id: command.session_id.clone(), + workflow_run_id: command.workflow_run_id.clone(), + response, + diagnostics, + }, + )) } pub(super) fn runtime_branch_failed( @@ -1369,14 +1371,14 @@ async fn claim_and_execute_runtime_branch_event( ) .await { - Ok(response) => WorkflowTaskExecutionWorkerOutcome::RuntimeBranchCompleted( - Box::new(WorkflowTaskExecutionWorkerRuntimeBranchCompletedOutcome { + Ok(response) => WorkflowTaskExecutionWorkerOutcome::RuntimeBranchCompleted(Box::new( + WorkflowTaskExecutionWorkerRuntimeBranchCompletedOutcome { session_id: command.session_id.clone(), workflow_run_id: command.workflow_run_id.clone(), response, diagnostics: Vec::new(), - }), - ), + }, + )), Err(error) => WorkflowTaskExecutionWorkerOutcome::runtime_branch_failed( command, error.to_string(), diff --git a/crates/pantograph-workflow-service/src/workflow/tests/task_binding_resolution.rs b/crates/pantograph-workflow-service/src/workflow/tests/task_binding_resolution.rs index 186d08109..18f14a6a8 100644 --- a/crates/pantograph-workflow-service/src/workflow/tests/task_binding_resolution.rs +++ b/crates/pantograph-workflow-service/src/workflow/tests/task_binding_resolution.rs @@ -33,8 +33,8 @@ fn workflow_run_id() -> WorkflowRunId { fn inference_projection() -> WorkflowSchedulerInferenceTaskProjections { WorkflowSchedulerInferenceTaskProjections::from_records(vec![ - WorkflowSchedulerInferenceTaskProjection::Ready( - Box::new(WorkflowSchedulerReadyInferenceTaskProjection { + WorkflowSchedulerInferenceTaskProjection::Ready(Box::new( + WorkflowSchedulerReadyInferenceTaskProjection { node_id: SchedulerNodeId::parse("infer").expect("node id"), descriptor_fingerprint: InferenceInterfaceFingerprint::parse("iface.binding.v1") .expect("fingerprint"), @@ -50,8 +50,8 @@ fn inference_projection() -> WorkflowSchedulerInferenceTaskProjections { trait_settings: Vec::new(), estimate_hints: resource_estimate_hints(), dependency_readiness_source: dependency_readiness_source("iface.binding.v1"), - }), - ), + }, + )), ]) .expect("projection") } diff --git a/crates/pantograph-workflow-service/src/workflow/tests/task_graph.rs b/crates/pantograph-workflow-service/src/workflow/tests/task_graph.rs index ec9a18eaa..30d8d4611 100644 --- a/crates/pantograph-workflow-service/src/workflow/tests/task_graph.rs +++ b/crates/pantograph-workflow-service/src/workflow/tests/task_graph.rs @@ -44,8 +44,8 @@ fn ready_inference_projection( estimate_hints: Vec, ) -> WorkflowSchedulerInferenceTaskProjections { WorkflowSchedulerInferenceTaskProjections::from_records(vec![ - WorkflowSchedulerInferenceTaskProjection::Ready( - Box::new(WorkflowSchedulerReadyInferenceTaskProjection { + WorkflowSchedulerInferenceTaskProjection::Ready(Box::new( + WorkflowSchedulerReadyInferenceTaskProjection { node_id: SchedulerNodeId::parse("infer").expect("node id"), descriptor_fingerprint: InferenceInterfaceFingerprint::parse("iface.test.v1") .expect("fingerprint"), @@ -70,8 +70,8 @@ fn ready_inference_projection( }], estimate_hints, dependency_readiness_source: dependency_readiness_source("iface.test.v1"), - }), - ), + }, + )), ]) .expect("projection") } diff --git a/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-pinned-rustfmt-correction.md b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-pinned-rustfmt-correction.md new file mode 100644 index 000000000..6ae6c11db --- /dev/null +++ b/docs/plans/domain-architecture-and-multimodal/reports/2026-10-03-pinned-rustfmt-correction.md @@ -0,0 +1,21 @@ +# Pinned rustfmt correction for the reviewed repair stack + +The full PR #39 format audit failed on wrapping and indentation introduced by recent boxed constructors and terminal calls. Reproduced locally with the repository-pinned Rust 1.92.0 toolchain: cargo 1.92.0 (344c4567c 2025-10-21), rustfmt 1.8.0-stable (ded5c06cf2 2025-12-08). Ran the actual `cargo fmt --all -- --check` before and after, rather than treating focused syntax/whitespace checks as format qualification. + +Formatted only the eleven files named by the pre-check (including the subsequent registry test assertion). No other baseline files were changed. The full post-check exits zero. A Rust lexer comparison preserves every non-whitespace token after ignoring optional trailing commas before closing delimiters; string/comment tokens remain exact. The registry test file has exact non-whitespace token equality. These are formatter changes, with no intended behavior or public API change; hosted aggregate qualification remains required. + +Changed source files: + +- `crates/pantograph-diagnostics-ledger/src/tests.rs` +- `crates/pantograph-embedded-runtime/src/node_execution_ledger.rs` +- `crates/pantograph-runtime-registry/src/technical_fit_tests.rs` +- `crates/pantograph-workflow-service/src/graph/inference_validation_state.rs` +- `crates/pantograph-workflow-service/src/scheduler/task_orchestrator.rs` +- `crates/pantograph-workflow-service/src/workflow/executable_validation_snapshot.rs` +- `crates/pantograph-workflow-service/src/workflow/session_execution_api.rs` +- `crates/pantograph-workflow-service/src/workflow/session_scheduler_runner.rs` +- `crates/pantograph-workflow-service/src/workflow/task_execution_worker.rs` +- `crates/pantograph-workflow-service/src/workflow/tests/task_binding_resolution.rs` +- `crates/pantograph-workflow-service/src/workflow/tests/task_graph.rs` + +The accepted registry commit c369cbe4b12b563eb8a5d4419e6e0ed729721137 and PR #39 history remain preserved. Root source review accepted frozen tree 62f3f04d4017d29592cc243167ae46841bcaa901 after checking the file list, pinned formatter/token receipts and representative full constructor/call diffs. The earlier red hosted format runs remain recorded; actual full hosted format and all other aggregate gates still require qualification.