33 Commits

Author SHA1 Message Date
77ae132e56 Prepare the v0.6.0 release 2026-08-29 12:46:18 +00:00
deebc89255 Make command line references pipeline scoped 2026-08-29 12:34:31 +00:00
f208dbe954 Make combat scene validation more reliable 2026-08-29 11:39:02 +00:00
7e626753bf Clarify semantic reconciliation candidate numbers 2026-08-29 03:16:24 +00:00
917d150279 Improve D&D validation reliability 2026-08-29 01:24:45 +00:00
4da9360d74 Improve semantic reconciliation retries 2026-08-28 18:25:00 +00:00
5cab4e512e Improve item occurrence holder corrections 2026-08-28 14:37:22 +00:00
b178f1c684 Repair reversed D&D evidence ranges 2026-08-28 13:45:21 +00:00
a2610757cd Polish combat scene validation 2026-08-28 02:26:14 +00:00
3ad34956c5 Complete combat semantics validator verification 2026-08-28 00:24:35 +00:00
22e6caa2a0 Document combat semantics validator evaluation 2026-08-28 00:23:50 +00:00
9cb7462800 Test combat semantics correction flow 2026-08-28 00:22:56 +00:00
87385b7e14 Register optional combat semantics validator 2026-08-28 00:20:59 +00:00
0ae5ea7637 Implement combat semantics validator 2026-08-28 00:19:49 +00:00
6bc883dfb6 Add combat semantics validator prompt assets 2026-08-28 00:16:23 +00:00
fcfff3ad15 Extract shared D&D combat policy 2026-08-28 00:13:44 +00:00
0f2b23dce1 Plan combat scene validation 2026-08-27 23:26:59 +00:00
61436d7c18 Prepare the v0.5.0 release 2026-08-27 19:05:13 +00:00
079d5af337 Harden diagnostic handling and warning presentation 2026-08-27 18:41:59 +00:00
1025001f20 Complete diagnostic migration verification 2026-08-27 16:54:05 +00:00
da14924a02 Document diagnostic ownership and classification 2026-08-27 16:52:58 +00:00
2065a8288b Align debug summaries with diagnostics 2026-08-27 16:51:56 +00:00
af0119cc1d Add diagnostic counts to run receipts 2026-08-27 16:50:32 +00:00
54de2b816a Publish grouped warning and diagnostic files 2026-08-27 16:43:09 +00:00
4dbbf68051 Finalize structured diagnostic aggregation 2026-08-27 16:40:17 +00:00
480680b257 Classify remaining D&D producer diagnostics 2026-08-27 16:04:32 +00:00
6a1fd7bdb6 Classify registry normalization diagnostics 2026-08-27 15:57:29 +00:00
ccba2ce3f9 Classify source relatedness as data-quality advisories 2026-08-27 15:49:53 +00:00
1f1967c8d2 Persist diagnostics in reusable pipeline state 2026-08-27 15:40:45 +00:00
5175cb0722 Classify framework process diagnostics 2026-08-27 15:33:13 +00:00
ba569594a1 Transport structured diagnostics through the pipeline 2026-08-27 15:29:10 +00:00
acb04954eb Add diagnostic contract primitives 2026-08-27 15:19:56 +00:00
610dd3d7c3 Plan warning and diagnostic reform 2026-08-27 15:06:43 +00:00
269 changed files with 8251 additions and 2762 deletions

View File

@@ -16,7 +16,8 @@
"type": "string" "type": "string"
}, },
"turn_kind": { "turn_kind": {
"type": "string" "type": "string",
"enum": ["turn", "reaction", "legendary_action", "lair_action", "other"]
}, },
"source_refs": { "source_refs": {
"type": "array", "type": "array",

View File

@@ -13,7 +13,10 @@
"required": ["name", "kind", "source_refs"], "required": ["name", "kind", "source_refs"],
"properties": { "properties": {
"name": {"type": "string"}, "name": {"type": "string"},
"kind": {"type": "string"}, "kind": {
"type": "string",
"enum": ["engaged", "killed", "fled", "captured", "incapacitated"]
},
"source_refs": { "source_refs": {
"type": "array", "type": "array",
"items": { "items": {

View File

@@ -18,12 +18,20 @@ when the transcript explicitly describes it being physically destroyed or
expended as a non-payment component. Use `transferred` only when possession expended as a non-payment component. Use `transferred` only when possession
moves between two distinct named party members. moves between two distinct named party members.
Return both `from` and `to` for every occurrence, using `null` when a holder does not Return both `from` and `to` for every occurrence. Use JSON `null`, not an empty
apply. For `discovered`, set both holders to `null`. For `acquired`, set `from` string, whenever a holder does not apply. Follow this holder matrix exactly:
to `null` and provide `to`; for `lost` and `consumed`, provide `from` and set
`to` to `null`; and for `transferred`, provide both holders. Use `party` only | `kind` | required `from` | required `to` |
for collective or unresolved party possession, never for either side of a | --- | --- | --- |
transfer. Do not emit a transfer for a gift, sale, or payment outside the party. | `discovered` | `null` | `null` |
| `acquired` | `null` | `party` or the named party member gaining possession |
| `lost` | `party` or the named party member losing possession | `null` |
| `consumed` | `party` or the named party member consuming the item | `null` |
| `transferred` | one named party member | a different named party member |
Use `party` only for collective or unresolved party possession, never for
either side of a transfer. Do not emit a transfer for a gift, sale, or payment
outside the party.
Ordinary non-depleting use is not an occurrence. Do not infer acquisition from a Ordinary non-depleting use is not an occurrence. Do not infer acquisition from a
discovery, or discovery from an acquisition: emit both only when each is discovery, or discovery from an acquisition: emit both only when each is

View File

@@ -13,7 +13,10 @@
"required": ["name", "kind", "quantity", "from", "to", "source_refs"], "required": ["name", "kind", "quantity", "from", "to", "source_refs"],
"properties": { "properties": {
"name": {"type": "string"}, "name": {"type": "string"},
"kind": {"type": "string"}, "kind": {
"type": "string",
"enum": ["discovered", "acquired", "lost", "consumed", "transferred"]
},
"quantity": {"type": ["integer", "null"]}, "quantity": {"type": ["integer", "null"]},
"from": {"type": ["string", "null"]}, "from": {"type": ["string", "null"]},
"to": {"type": ["string", "null"]}, "to": {"type": ["string", "null"]},

View File

@@ -5,5 +5,5 @@ evidence, similar objects, or a shared owner as sufficient.
Keep currency denominations and materially different item types separate. Keep Keep currency denominations and materially different item types separate. Keep
uncertain aliases separate. Do not infer an item property or uniqueness. uncertain aliases separate. Do not infer an item property or uniqueness.
When selecting a canonical display name, choose one supplied candidate name Set `canonical_candidate_number` to the supplied candidate number whose label
that is the clearest established designation. is the clearest established designation.

View File

@@ -5,4 +5,5 @@ nearby evidence, nested places, or generic labels as sufficient.
Keep parent and child places separate, as well as similarly named places and Keep parent and child places separate, as well as similarly named places and
uncertain aliases. uncertain aliases.
When selecting a canonical display name, prefer the clearest established name. Set `canonical_candidate_number` to the supplied candidate number whose label
is the clearest established name.

View File

@@ -16,7 +16,8 @@
"type": "string" "type": "string"
}, },
"kind": { "kind": {
"type": "string" "type": "string",
"enum": ["mentioned", "noncombat_presence", "dialogue", "combat_ally", "combat_opponent", "other"]
}, },
"source_refs": { "source_refs": {
"type": "array", "type": "array",

View File

@@ -3,7 +3,8 @@ contextual labels and cited transcript windows. Preserve distinct individuals
even when their names are similar or their contextual descriptions are even when their names are similar or their contextual descriptions are
identical. identical.
When selecting a canonical display name, prefer a complete, stable proper name Set `canonical_candidate_number` to the supplied candidate number whose label
is the preferred canonical display name. Prefer a complete, stable proper name
over an abbreviation. Prefer an unadorned proper name over that name plus a over an abbreviation. Prefer an unadorned proper name over that name plus a
contextual class, role, title, or relationship descriptor unless the transcript contextual class, role, title, or relationship descriptor unless the transcript
establishes the descriptor as part of the person's name. A longer display name establishes the descriptor as part of the person's name. A longer display name

View File

@@ -5,9 +5,7 @@ into multiple scenes or use facts that are not supported by it.
Return one kind, one concise title, and one concise summary. Choose exactly one Return one kind, one concise title, and one concise summary. Choose exactly one
kind: kind:
- combat: active combat materially organizes the scene, including - combat: a scene classified as combat under the shared combat policy.
initiative-like exchanges or sustained hostile action. Planning a fight or
discussing a completed fight is not combat by itself.
- narrative: current-session in-world play that is not principally active - narrative: current-session in-world play that is not principally active
combat, a prior-session recap, or sustained out-of-character session combat, a prior-session recap, or sustained out-of-character session
discussion. This includes exploration, travel, dialogue, investigation, discussion. This includes exploration, travel, dialogue, investigation,
@@ -20,15 +18,13 @@ kind:
play. play.
Narrative is the default for actual current-session gameplay that does not meet Narrative is the default for actual current-session gameplay that does not meet
another definition. When the accepted chunk is mixed: another definition. When the accepted chunk has no substantive active combat:
1. use combat when active combat is a substantive central activity, even with 1. use recap when recounting a previous session is the chunk's
brief setup, rules clarification, or immediate aftermath;
2. otherwise use recap when recounting a previous session is the chunk's
primary table purpose; primary table purpose;
3. otherwise use meta when sustained out-of-character session discussion is 2. otherwise use meta when sustained out-of-character session discussion is
primary and in-world progression is no more than incidental; and primary and in-world progression is no more than incidental; and
4. use narrative for all remaining current-session in-world play. 3. use narrative for all remaining current-session in-world play.
Brief table talk, dice resolution, rules clarification, jokes, or Brief table talk, dice resolution, rules clarification, jokes, or
administrative comments do not make a gameplay scene meta. A short recollection administrative comments do not make a gameplay scene meta. A short recollection

View File

@@ -27,6 +27,8 @@ messages:
content_file: ./sharedassets/common-dnd-transcript-chunk.md content_file: ./sharedassets/common-dnd-transcript-chunk.md
cache_control: cache_control:
type: ephemeral type: ephemeral
- role: user
content_file: ./sharedassets/common-dnd-scene-combat-policy.md
- role: user - role: user
content_file: ./instructions.md content_file: ./instructions.md
cache_control: cache_control:

View File

@@ -6,7 +6,8 @@
"required": ["kind", "title", "summary"], "required": ["kind", "title", "summary"],
"properties": { "properties": {
"kind": { "kind": {
"type": "string" "type": "string",
"enum": ["combat", "narrative", "recap", "meta"]
}, },
"title": { "title": {
"type": "string" "type": "string"

View File

@@ -0,0 +1,8 @@
Classify only whether substantive active combat occurs in the supplied
transcript chunk under the shared combat policy. Do not judge the scene title,
summary, non-combat subtype, scene boundary, or any other aspect of a scene
description.
Return `combat` when the chunk contains substantive active combat and
`non_combat` otherwise. Give a concise, transcript-grounded explanation for
the classification.

View File

@@ -0,0 +1,23 @@
id: dnd.scene_descriptions.validate_combat
version: "v1"
default_profile: dnd-extraction
inputs:
- name: transcript
required: true
content_type: application/json
messages:
- role: system
content_file: ./sharedassets/common-dnd-system.md
- role: user
content_file: ./sharedassets/common-dnd-scene-combat-policy.md
- role: user
content_file: ./instructions.md
- role: user
content_file: ./sharedassets/common-dnd-transcript-chunk.md
cache_control:
type: ephemeral
output:
format: json
validation_mode: json_schema
schema_path: dnd_scene_combat_semantics_llm.v1.json
repair_attempts: 1

View File

@@ -0,0 +1,18 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "notarius.dnd.scene_descriptions.combat_semantics.llm",
"type": "object",
"additionalProperties": false,
"required": ["classification", "explanation"],
"properties": {
"classification": {
"type": "string",
"enum": ["combat", "non_combat"]
},
"explanation": {
"type": "string",
"minLength": 1,
"maxLength": 512
}
}
}

View File

@@ -1,6 +1,8 @@
Transcript units are the only evidence for extracted events and factual claims. Transcript units are the only evidence for extracted events and factual claims.
Every reported factual claim must be supported by cited transcript units. Use Every reported factual claim must be supported by cited transcript units. Use
integer `start_unit_id` and `end_unit_id` values from the transcript. integer `start_unit_id` and `end_unit_id` values from the transcript.
Within each range, `start_unit_id` must identify the earlier transcript unit and
`end_unit_id` the same or a later unit according to transcript order.
When supporting evidence is non-contiguous, use multiple narrow ranges rather When supporting evidence is non-contiguous, use multiple narrow ranges rather
than a broad range that bridges unrelated conversation. than a broad range that bridges unrelated conversation.

View File

@@ -0,0 +1,13 @@
Use `combat` only when substantive active combat materially organizes the
scene. Active combat includes initiative or turn exchanges, attacks, combat
spells, damage, saves, movement, or similarly sustained hostile action.
Do not use `combat` for planning or preparing for a possible fight; threats,
hostile dialogue, or a tense confrontation; immediate aftermath, looting,
healing, or discussion of a completed fight; a recap or in-world recollection
of earlier combat; or out-of-character rules discussion without active
encounter play.
When a chunk contains substantive active combat alongside brief setup, rules
clarification, interruption, phase transition, or immediate aftermath, classify
it as `combat`.

View File

@@ -1,3 +1,3 @@
Candidate material: Candidate material, including the exact valid candidate-number range:
{{ input "candidates" }} {{ input "candidates" }}

View File

@@ -1,7 +1,9 @@
Use only the positive integer `candidate_id` values supplied in the candidate material. Use only the positive integer `candidate_number` values supplied in the candidate material. Valid candidate numbers are exactly the inclusive `first` through `last` values declared in `candidate_number_range`; use the explicit number attached to each candidate.
Return a duplicate group only when the evidence supports that every selected candidate describes the same underlying entity. Each group must contain at least two distinct candidate IDs, and its `canonical_candidate_id` must be one of those IDs. A candidate may appear in at most one group. Transcript unit `id` values and evidence `start_unit_id` and `end_unit_id` values identify source positions. They are not candidate numbers and must never be used in `candidate_numbers` or `canonical_candidate_number`.
Omit uncertain matches and candidates that should remain distinct. Do not invent candidates or infer an ID from list position. An empty `duplicate_groups` array is valid. Return a duplicate group only when the evidence supports that every selected candidate describes the same underlying entity. Each group must contain at least two distinct candidate numbers, and its `canonical_candidate_number` must be one of those numbers. A candidate may appear in at most one group.
The response must conform exactly to the selected JSON schema. Return IDs only: do not copy candidate names, evidence, transcript text, source identifiers, or source ranges into the response. Omit uncertain matches and candidates that should remain distinct. Do not invent candidates or infer a number that is not explicitly supplied. An empty `duplicate_groups` array is valid.
The response must conform exactly to the selected JSON schema. Return candidate numbers only: do not copy candidate names, evidence, transcript text, source identifiers, or source ranges into the response.

View File

@@ -11,9 +11,9 @@
"items": { "items": {
"type": "object", "type": "object",
"additionalProperties": false, "additionalProperties": false,
"required": ["candidate_ids", "canonical_candidate_id"], "required": ["candidate_numbers", "canonical_candidate_number"],
"properties": { "properties": {
"candidate_ids": { "candidate_numbers": {
"type": "array", "type": "array",
"minItems": 2, "minItems": 2,
"items": { "items": {
@@ -21,7 +21,7 @@
"minimum": 1 "minimum": 1
} }
}, },
"canonical_candidate_id": { "canonical_candidate_number": {
"type": "integer", "type": "integer",
"minimum": 1 "minimum": 1
} }

View File

@@ -0,0 +1,67 @@
# ADR-0015: Separate process warnings from quality diagnostics
**Status:** Accepted
**Date:** 2026-08-27
## Context
Notarius currently represents process degradation, incomplete validation,
extraction-quality doubt, and routine normalization with one flat warning
record. That makes ordinary successful runs noisy, loses the framework context
needed to explain a finding, and gives `warning_count` no stable operational
meaning. It also permits output encoders to add a warning after the durable
warning file has already been written.
The application needs one bounded diagnostic model that preserves exact
occurrence counts while retaining only safe, representative samples. Fresh and
resumed logical runs must present the same groups. The model must not alter
validation decisions, retry budgets, rejected-output behavior, or process exit
policy.
## Decision
Warnings are reserved for a completed run that advanced under an allowed
process-level degradation or incomplete-work policy. Extraction-quality signals
are advisories, and routine accepted transformations are observations. A
non-degraded successful run therefore has zero actionable warnings.
Modules and validators own a diagnostic's disposition, category, reason code,
scope, and safe message. The framework adds pipeline origin, including stage,
step, lane, module, validator, and chunk context where applicable. It then
aggregates deterministically by disposition, category, reason code, and full
origin. Chunk context remains on representative samples so equivalent findings
across chunks aggregate together.
Diagnostics carry exact occurrence counts, at most three distinct samples, and
numeric omitted-sample metadata. Producers and validators are bounded to 64
local groups. Final actionable warning groups are bounded without truncation;
the non-warning collection may truncate represented groups while preserving an
exact total occurrence count and explicit truncation metadata.
The public contracts will be versioned: grouped actionable warnings use
`notarius.warnings.v2`, grouped advisories and observations use
`notarius.diagnostics.v1`, and the run receipt uses
`notarius.run-result.v2`. Successful output encoders return logical files or
an error; they do not add post-encoding warnings.
## Alternatives considered
- Keep one warning list and filter only CLI output. This would leave durable
consumers with the same semantically mixed, unbounded contract.
- Map reason codes to severity in a central framework registry. This would
split module-owned meaning between synchronized policy tables and make new
diagnostic meaning implicit.
- Preserve local omission warning records. They inflate visible group counts
and lose exact occurrence semantics.
- Keep output-encoder warnings. A one-pass encoder cannot include those
records consistently in files it has already serialized; a two-phase encoder
protocol is deferred until a demonstrated need exists.
## Consequences
The framework gains validated diagnostic primitives, local collection,
origin-aware aggregation, and versioned durable presentation. Existing warning
transport remains temporarily while producers migrate. Current architecture,
operator, integration, and internal documentation will describe the behavior
only as each implementation step lands; this accepted decision does not claim
that the migration is complete.

View File

@@ -0,0 +1,76 @@
# ADR-0016: Use feedback-aware module-requested retries
**Status:** Accepted
**Date:** 2026-08-28
## Context
An LLM-backed module can discover that a structurally valid model response is
unsafe while translating it into a typed candidate, before the ordinary
validator chain runs. Semantic registry reconciliation is the first such case:
the shared core can identify invalid duplicate-group proposals, and typed
application can reject a domain-incompatible group while preserving a safe
partial result. Repeating the original request without the rejected response or
corrective explanation gives the model no information with which to improve.
The existing feedback-aware validation mechanism already establishes the safe
correction protocol, but module-owned retry directives also carry internal
reason codes, operator messages, and fallback diagnostics. Those values are not
model instructions. Some module retry conditions, including exhausted
structured-output repair, also have no valid exact response to append.
## Decision
An LLM-backed normalizer may attach optional, bounded model-facing correction
guidance to a module-requested retry. Guidance is a separate contract field; the
framework never derives it from a reason code, operator message, diagnostic, or
error. A feedback-capable directive must include the exact model candidate that
controlled the safe fallback and must use `single_response_v1`.
The artifact-neutral producer-attempt state machine constructs the correction
from that exact latest response and the supplied guidance. The producer rebuilds
its complete ordinary request and appends the response as one assistant message
and the guidance as one user message. Earlier attempts do not accumulate, and
the attempt consumes the same configured stage retry budget as operational,
structural, validation, and feedback-free module retries.
A module retry without guidance remains valid and starts a fresh attempt. This
is the required behavior when no valid model candidate exists, including after
structured-output repair is exhausted. If feedback is supplied without a valid
supported candidate, the framework fails the module contract rather than
retrying blindly or inventing guidance.
After retry exhaustion, the normalizer's safe fallback continues through its
validator chain. Safe semantic groups may remain applied, unsafe groups remain
separate, and bounded fallback diagnostics may explain the process degradation.
Exact responses and correction text remain attempt-local and are excluded from
ordinary errors, warnings, manifests, receipts, caches, and checkpoints.
This decision extends, rather than supersedes,
[ADR-0014](0014-feedback-aware-validation-retries.md): both correction sources
use the same transport-neutral payload, replacement-request semantics, outer
retry budget, and sensitive-data boundary.
## Alternatives considered
- Continue blind module retries. This preserves a smaller contract but wastes
the module's deterministic diagnosis and commonly repeats the same defect.
- Convert module safety checks into validators. Typed reconciliation must apply
only safe proposal groups and retain a fallback before validation; moving
artifact-owned translation and application policy into validators would blur
stage ownership.
- Copy the retry reason or operator message into the model request. Those values
are written for provenance and humans, can contain opaque internal labels,
and do not reliably describe a correct replacement.
- Require feedback for every module retry. Structural failures may have no
valid exact candidate, so this would either prevent useful fresh retries or
fabricate prior-response material.
## Consequences
The normalize retry contract and generic producer-attempt directive gain an
optional correction-guidance field and candidate-pairing validation. Modules
that use it must provide semantically meaningful bounded prose and an exact
candidate. Registry reconciliation maintains separate operator and model
renderers, and policy fingerprints change so checkpoints created under blind
retry behavior are not reused.

View File

@@ -55,8 +55,8 @@ pipeline ID and **--input** are required.
| **--session-id id** | Override the generated prompt session identifier with a non-empty value for LLM-backed module calls. | | **--session-id id** | Override the generated prompt session identifier with a non-empty value for LLM-backed module calls. |
| **--reasoning-effort value** | Replace the selected PromptKit profile's reasoning effort for every LLM-backed call in this run. The value must be non-empty and the flag may be specified only once. | | **--reasoning-effort value** | Replace the selected PromptKit profile's reasoning effort for every LLM-backed call in this run. The value must be non-empty and the flag may be specified only once. |
| **--clear-reasoning-effort** | Clear reasoning effort inherited from the selected PromptKit profile for every LLM-backed call in this run. | | **--clear-reasoning-effort** | Clear reasoning effort inherited from the selected PromptKit profile for every LLM-backed call in this run. |
| **--reference selector=path** | Add or replace a file reference binding. Repeatable. | | **--reference selector=path** | Add or replace external file reference bindings at pipeline, lane, chunk, or binding scope. Repeatable. |
| **--without-reference selector** | Remove a configured optional reference binding. Repeatable. | | **--without-reference selector** | Remove matching configured external reference bindings. Repeatable. |
**--chunk_cache** accepts only **auto**, **bypass**, or **refresh**. **--chunk_cache** accepts only **auto**, **bypass**, or **refresh**.
**--debug-dir**, **--output-dir**, **--session-id**, and **--debug-dir**, **--output-dir**, **--session-id**, and
@@ -81,33 +81,62 @@ guidance.
### Reference selectors ### Reference selectors
Use **--reference** only for a reference slot declared by the selected Use **--reference** only for reference slots declared by the selected
configured target. The accepted selector forms are: configured targets. Qualification narrows the scope of an override:
| Form | Target | | Form | Target |
| --- | --- | | --- | --- |
| slot=path | The unique selected target that declares slot. | | slot=path | Every selected target that declares slot. |
| chunk.slot=path | The chunker. | | chunk.slot=path | The chunker. |
| merge.slot=path | The unique selected merger that declares slot. | | lane.slot=path | Every extractor, merger, or normalizer in lane that declares slot. |
| lane.slot=path | The unique extractor, merger, or normalizer in lane that declares slot. |
| lane.extract.slot=path | The extractor in lane. | | lane.extract.slot=path | The extractor in lane. |
| lane.merge.slot=path | The merger in lane. | | lane.merge.slot=path | The merger in lane. |
| lane.normalize.slot=path | The normalizer in lane. | | lane.normalize.slot=path | The normalizer in lane. |
**--without-reference** uses the same selector forms without =path. Slot Pipeline- and lane-scoped selectors are expected to match multiple targets and
names, requiredness, and configured bindings are part of the fail if they match none. A stage-specific selector fails when its lane is not
selected or its target does not declare the slot. There is no stage-wide
`merge.slot` shorthand; name the lane when targeting a merger.
CLI bindings override configured external paths. For overlapping CLI
selectors, a binding-specific or chunk selector overrides a lane selector, and
a lane selector overrides a pipeline selector. The last occurrence wins at
equal scope. Binding and unbinding the same concrete target at equal scope is
an error; a narrower bind or unbind may create an intentional exception to a
broader action.
**--without-reference** uses the same selector forms without `=path` and
removes external bindings only. Neither flag replaces or removes a generated
artifact handoff; an external/generated collision is a resolution error.
Required slots are checked after all effective changes. CLI reference paths
are resolved relative to the process working directory, so subprocess and
service callers should use absolute paths. Slot names, accepted media types,
size limits, requiredness, and configured generated bindings are part of the
[configuration contract](config.md). [configuration contract](config.md).
For example, one shared campaign reference can reach every compatible target,
with an optional lane-specific exception:
~~~
notarius run dnd-session \
--input /data/transcript.json \
--reference party=/data/references/party.txt \
--reference npc-registry.party=/data/references/npc-party-context.txt
~~~
### Run output ### Run output
Without **--json**, standard output contains the completed pipeline ID, counts Without **--json**, standard output contains the completed pipeline ID, counts
of normalized and rejected outputs, and the output directory. A debug-enabled of normalized and rejected outputs, and the output directory. A debug-enabled
run also prints its debug-bundle path to standard output. A successful run with run also prints its debug-bundle path to standard output. A successful run with
warnings reports the warning count to standard error. The published JSON bundle actionable process warnings reports their group and occurrence counts to
standard error. When the selected output module publishes `warnings.json`, the
summary also reports that durable file's path. Advisory and observation findings
do not produce a warning line. The published JSON bundle
is defined by the [JSON output contract](integrations/json-output.md). is defined by the [JSON output contract](integrations/json-output.md).
With **--json**, successful standard output is exactly one With **--json**, successful standard output is exactly one
`notarius.run-result.v1` JSON document followed by a newline, with no `notarius.run-result.v2` JSON document followed by a newline, with no
human-oriented status or debug-path line. Its fields and compatibility policy human-oriented status or debug-path line. Its fields and compatibility policy
are defined by the [run-result contract](integrations/run-result.md). A caller are defined by the [run-result contract](integrations/run-result.md). A caller
must check for exit status 0 before decoding this output; a failed write can must check for exit status 0 before decoding this output; a failed write can
@@ -171,7 +200,7 @@ go run ./cmd/notarius pipelines list \
Successful commands write their primary result to standard output. Warnings and Successful commands write their primary result to standard output. Warnings and
errors are written to standard error. errors are written to standard error.
For **run --json**, warnings remain on standard error and standard output is a For **run --json**, actionable process warnings remain on standard error and standard output is a
machine-readable success result only. Syntax and runtime diagnostics remain on machine-readable success result only. Syntax and runtime diagnostics remain on
standard error. Parse the result only after the process exits with status 0. standard error. Parse the result only after the process exits with status 0.

View File

@@ -411,10 +411,15 @@ slot. A generated binding supplies one accepted normalized artifact; it does
not name a file. A configured generated dependency remains required even when not name a file. A configured generated dependency remains required even when
that consumer slot is otherwise optional. that consumer slot is otherwise optional.
Pipeline references are defaults. A matching step-local or binding-local Pipeline references are configuration defaults. A matching step-local or
external path overrides a pipeline default. Required slots must be bound after binding-local external path overrides a pipeline default. CLI reference
these configuration values and any CLI reference overrides are applied. bindings are then operational overrides of configured external paths; their
Reference paths in YAML are resolved relative to the configuration file. pipeline, lane, and binding scopes and precedence are defined by the
[CLI reference](cli.md#reference-selectors). A CLI file reference cannot
replace a configured generated artifact handoff. Required slots must be bound
after configuration and CLI reference actions are applied. Reference paths in
YAML are resolved relative to the configuration file; CLI reference paths are
resolved relative to the process working directory.
### D&D Reference Slots ### D&D Reference Slots
@@ -504,11 +509,33 @@ Available validator keys are:
| Item occurrences | **extract/dnd/item-occurrences/shape**, **extract/dnd/item-occurrences/registry**, **extract/dnd/item-occurrences/source_refs**, **extract/dnd/item-occurrences/source_relatedness**, **normalize/dnd/item-occurrences/invariants** | | Item occurrences | **extract/dnd/item-occurrences/shape**, **extract/dnd/item-occurrences/registry**, **extract/dnd/item-occurrences/source_refs**, **extract/dnd/item-occurrences/source_relatedness**, **normalize/dnd/item-occurrences/invariants** |
| Item registry | **extract/dnd/item-registry/shape**, **extract/dnd/item-registry/source_refs**, **extract/dnd/item-registry/source_relatedness**, **normalize/dnd/item-registry/identity** | | Item registry | **extract/dnd/item-registry/shape**, **extract/dnd/item-registry/source_refs**, **extract/dnd/item-registry/source_relatedness**, **normalize/dnd/item-registry/identity** |
| NPC occurrences | **extract/dnd/npc-occurrences/shape**, **extract/dnd/npc-occurrences/registry**, **extract/dnd/npc-occurrences/source_refs**, **extract/dnd/npc-occurrences/source_relatedness**, **normalize/dnd/npc-occurrences/invariants** | | NPC occurrences | **extract/dnd/npc-occurrences/shape**, **extract/dnd/npc-occurrences/registry**, **extract/dnd/npc-occurrences/source_refs**, **extract/dnd/npc-occurrences/source_relatedness**, **normalize/dnd/npc-occurrences/invariants** |
| Scene descriptions | **extract/dnd/scene-descriptions/shape**, **extract/dnd/scene-descriptions/source_refs**, **extract/dnd/scene-descriptions/source_relatedness**, **normalize/dnd/scene-descriptions/invariants** | | Scene descriptions | **extract/dnd/scene-descriptions/shape**, **extract/dnd/scene-descriptions/source_refs**, **extract/dnd/scene-descriptions/source_relatedness**, **extract/dnd/scene-descriptions/combat_semantics** (LLM-backed, opt-in), **normalize/dnd/scene-descriptions/invariants** |
| Enemy events | **extract/dnd/enemy-events/shape**, **extract/dnd/enemy-events/engagements**, **extract/dnd/enemy-events/source_refs**, **extract/dnd/enemy-events/source_relatedness**, **normalize/dnd/enemy-events/invariants** | | Enemy events | **extract/dnd/enemy-events/shape**, **extract/dnd/enemy-events/engagements**, **extract/dnd/enemy-events/source_refs**, **extract/dnd/enemy-events/source_relatedness**, **normalize/dnd/enemy-events/invariants** |
| Location registry | **extract/dnd/location-registry/shape**, **extract/dnd/location-registry/source_refs**, **extract/dnd/location-registry/source_relatedness**, **normalize/dnd/location-registry/identity** | | Location registry | **extract/dnd/location-registry/shape**, **extract/dnd/location-registry/source_refs**, **extract/dnd/location-registry/source_relatedness**, **normalize/dnd/location-registry/identity** |
| Location occurrences | **extract/dnd/location-occurrences/shape**, **extract/dnd/location-occurrences/registry**, **extract/dnd/location-occurrences/source_refs**, **extract/dnd/location-occurrences/source_relatedness**, **normalize/dnd/location-occurrences/invariants** | | Location occurrences | **extract/dnd/location-occurrences/shape**, **extract/dnd/location-occurrences/registry**, **extract/dnd/location-occurrences/source_refs**, **extract/dnd/location-occurrences/source_relatedness**, **normalize/dnd/location-occurrences/invariants** |
`extract/dnd/scene-descriptions/combat_semantics` is not in a production default chain. To opt in, replace the scene extractor validator chain with the current ordered chain plus the semantic validator last, and set a positive producer retry budget if a rejection should request a corrected scene:
~~~yaml
extract:
module: dnd/scene-descriptions
retries: 1
validators:
- generic/valid_json
- extract/dnd/scene-descriptions/shape
- extract/dnd/scene-descriptions/source_refs
- generic/valid_json_schema
- extract/dnd/scene-descriptions/source_relatedness
- module: extract/dnd/scene-descriptions/combat_semantics
retries: 1
~~~
An override replaces, rather than extends, the default chain. See [Module Bindings And Validators](#module-bindings-and-validators) for binding, profile, repair, retry, and failure-policy rules.
The validator retry shown above permits one additional execution against the
same scene candidate when the LLM-backed validator itself fails; it is separate
from both the extractor's producer retry and PromptKit structural repair.
When no override is configured, production D&D bindings use the following When no override is configured, production D&D bindings use the following
ordered chains. Each row lists extract then normalize; spell chains are the ordered chains. Each row lists extract then normalize; spell chains are the
same at both stages. same at both stages.

View File

@@ -24,7 +24,9 @@ files. Use absolute paths for service and subprocess deployments. In
particular, observe these different resolution rules: particular, observe these different resolution rules:
- reference paths in YAML are resolved relative to the Notarius configuration - reference paths in YAML are resolved relative to the Notarius configuration
file; and file;
- reference paths passed with `--reference` are resolved relative to the
Notarius process working directory; and
- `promptkit.profile_file` is resolved relative to the Notarius process working - `promptkit.profile_file` is resolved relative to the Notarius process working
directory. directory.
@@ -73,9 +75,21 @@ notarius run dnd-session \
--config /absolute/path/to/notarius.yml \ --config /absolute/path/to/notarius.yml \
--input /absolute/path/to/transcripts/final.trimmed.json \ --input /absolute/path/to/transcripts/final.trimmed.json \
--output-dir /absolute/path/to/notarius-output \ --output-dir /absolute/path/to/notarius-output \
--reference party=/absolute/path/to/references/party.txt \
--reference players=/absolute/path/to/references/players.txt \
--reference glossary=/absolute/path/to/references/glossary.txt \
--reference spell_catalog=/absolute/path/to/references/spells.json \
--json --json
``` ```
Each unqualified reference is pipeline-scoped: Notarius supplies it to every
selected D&D target that declares the slot. A deployment may omit an optional
reference it does not maintain, and may use the lane- or binding-qualified
forms from the [CLI reference](../cli.md#reference-selectors) for an exceptional
override. The registry, scene-description, combat-turn, and NPC-occurrence
references declared between ordered steps in the complete configuration are
generated artifacts. Do not pass those handoffs on the CLI.
The caller should: The caller should:
- capture stdout and stderr separately; - capture stdout and stderr separately;
@@ -92,7 +106,7 @@ stream and exit-status contract.
## Discover The Published Bundle ## Discover The Published Bundle
Decode the successful stdout document as a supported run-result schema. For Decode the successful stdout document as a supported run-result schema. For
the current contract, `schema_version` is `notarius.run-result.v1`. Tolerate the current contract, `schema_version` is `notarius.run-result.v2`. Tolerate
unknown fields allowed by that version, but reject an unsupported schema unknown fields allowed by that version, but reject an unsupported schema
version. version.
@@ -145,7 +159,8 @@ The JSON encoder always publishes these bundle-management files:
| `index.json` | Discovery document for lane and pipeline-wide artifacts. | | `index.json` | Discovery document for lane and pipeline-wide artifacts. |
| `manifest.json` | Run provenance and result summaries. | | `manifest.json` | Run provenance and result summaries. |
| `rejected.json` | Rejected pipeline outputs. | | `rejected.json` | Rejected pipeline outputs. |
| `warnings.json` | Accepted-output and run warnings. | | `warnings.json` | Actionable process-degradation warnings. |
| `diagnostics.json` | Advisory and observation findings for accepted artifacts. |
The complete configuration also requests two pipeline-wide artifacts: The complete configuration also requests two pipeline-wide artifacts:

View File

@@ -31,13 +31,36 @@ notarius run pipeline-id \
``` ```
Use absolute paths for supplied input, configuration, output-root, and Use absolute paths for supplied input, configuration, output-root, and
reference files. Notarius generates a stable prompt session for the resolved reference files. Pass each external reference as its own argument-vector pair;
input module and exact input bytes. Pass **--session-id** only when intentionally do not construct and invoke a shell command. An unqualified reference selector
grouping different invocations under a different session. Supply credentials supplies that file to every compatible selected target. Lane and stage
through Notarius's documented configuration and environment mechanisms, never qualification are available for exceptional overrides, while generated
as command-line arguments or generated secret-bearing configuration. In same-run references remain part of configured pipeline composition. The
particular, a session identifier is provider-visible and is not a credential [CLI reference](../cli.md#reference-selectors) owns the exact selector and
mechanism. precedence contract.
The maintained D&D subprocess workflow uses this facility for campaign context:
```sh
notarius run dnd-session \
--config /absolute/path/to/notarius.yml \
--input /absolute/path/to/transcripts/final.trimmed.json \
--output-dir /absolute/path/to/notarius-output \
--reference party=/absolute/path/to/references/party.txt \
--reference players=/absolute/path/to/references/players.txt \
--reference glossary=/absolute/path/to/references/glossary.txt \
--reference spell_catalog=/absolute/path/to/references/spells.json \
--json
```
Only pass the external references available to and desired by the deployment.
Notarius generates a stable prompt session for the resolved input module and
exact input bytes; reference changes do not change it. Pass **--session-id**
only when intentionally grouping different invocations under a different
session. Supply credentials through Notarius's documented configuration and
environment mechanisms, never as command-line arguments or generated
secret-bearing configuration. In particular, a session identifier is
provider-visible and is not a credential mechanism.
Wait for the process before interpreting standard output. Only an exit status Wait for the process before interpreting standard output. Only an exit status
of 0 permits decoding the receipt. On a nonzero exit, retain standard error for of 0 permits decoding the receipt. On a nonzero exit, retain standard error for
@@ -79,7 +102,7 @@ silently treated as fully reviewed by the caller.
## Preserve Provenance And Handle Data Carefully ## Preserve Provenance And Handle Data Carefully
Keep the receipt with the published `manifest.json`, and retain Keep the receipt with the published `manifest.json`, and retain
`rejected.json` and `warnings.json` when review or later provenance requires `rejected.json`, `warnings.json`, and `diagnostics.json` when review or later provenance requires
them. Treat the input, output bundle, cache, debug bundle, and captured process them. Treat the input, output bundle, cache, debug bundle, and captured process
logs as potentially sensitive data. Apply the caller's access controls and logs as potentially sensitive data. Apply the caller's access controls and
retention policy, and avoid copying secrets into arguments, logs, or retention policy, and avoid copying secrets into arguments, logs, or

View File

@@ -9,7 +9,7 @@ Output configuration, including chunk-map and evidence-context publication, belo
## Bundle Layout ## Bundle Layout
All paths below are logical, relative, slash-separated bundle paths. The All paths below are logical, relative, slash-separated bundle paths. The
encoder always emits the first four JSON files below and adds lane or encoder always emits the first five JSON files below and adds lane or
pipeline-wide artifact files when their corresponding artifacts are available: pipeline-wide artifact files when their corresponding artifacts are available:
A subprocess caller first obtains the physical bundle root from the A subprocess caller first obtains the physical bundle root from the
@@ -21,7 +21,8 @@ root for the logical discovery described here.
| `index.json` | Entry point that names the other published files and lane payloads. | | `index.json` | Entry point that names the other published files and lane payloads. |
| `manifest.json` | Run provenance and result summaries. | | `manifest.json` | Run provenance and result summaries. |
| `rejected.json` | Rejected pipeline outputs. | | `rejected.json` | Rejected pipeline outputs. |
| `warnings.json` | Accepted-output and run warnings. | | `warnings.json` | Actionable process-degradation warnings. |
| `diagnostics.json` | Accepted-artifact quality advisories and normalization observations. |
| `lanes/<safe-lane-id>.json` | One normalized artifact payload for each lane. | | `lanes/<safe-lane-id>.json` | One normalized artifact payload for each lane. |
| `chunk-map.json` | Optional accepted chunk map, when its export is enabled and available. | | `chunk-map.json` | Optional accepted chunk map, when its export is enabled and available. |
| `evidence-context.json` | Optional selected source-unit excerpt, when evidence publication is enabled. | | `evidence-context.json` | Optional selected source-unit excerpt, when evidence publication is enabled. |
@@ -39,7 +40,8 @@ normalized lanes has this valid minimal index:
"manifest_file": "manifest.json", "manifest_file": "manifest.json",
"output_files": [], "output_files": [],
"rejected_file": "rejected.json", "rejected_file": "rejected.json",
"warnings_file": "warnings.json" "warnings_file": "warnings.json",
"diagnostics_file": "diagnostics.json"
} }
``` ```
@@ -49,6 +51,7 @@ normalized lanes has this valid minimal index:
| `output_files` | Yes | Lane descriptors sorted by `lane_id`. | | `output_files` | Yes | Lane descriptors sorted by `lane_id`. |
| `rejected_file` | Yes | Always `rejected.json`. | | `rejected_file` | Yes | Always `rejected.json`. |
| `warnings_file` | Yes | Always `warnings.json`. | | `warnings_file` | Yes | Always `warnings.json`. |
| `diagnostics_file` | Yes | Always `diagnostics.json`. |
| `chunk_map` | No | Descriptor for the pipeline-wide `chunk-map.json`; never a lane descriptor. | | `chunk_map` | No | Descriptor for the pipeline-wide `chunk-map.json`; never a lane descriptor. |
| `evidence_context` | No | Descriptor for the pipeline-wide `evidence-context.json`; never a lane descriptor. | | `evidence_context` | No | Descriptor for the pipeline-wide `evidence-context.json`; never a lane descriptor. |
@@ -132,7 +135,7 @@ These values describe observed execution; they are not a backend-registration
interface. Entries that differ by backend or effective reasoning remain interface. Entries that differ by backend or effective reasoning remain
distinct even when their profile, provider, and model are otherwise equal. distinct even when their profile, provider, and model are otherwise equal.
## Rejections And Warnings ## Rejections, Warnings, And Diagnostics
`rejected.json` is always an object with a `rejected` array. Each entry has `rejected.json` is always an object with a `rejected` array. Each entry has
required `stage` and `message`; `step_id`, `lane_id`, `module_key`, `chunk_id`, required `stage` and `message`; `step_id`, `lane_id`, `module_key`, `chunk_id`,
@@ -142,9 +145,47 @@ contain the bounded `validation` summary described above; the existing singular
validator and reason fields remain the first configured rejection for validator and reason fields remain the first configured rejection for
compatibility. compatibility.
`warnings.json` is always an object with a `warnings` array. Each warning has `warnings.json` is always the `notarius.warnings.v2` envelope:
`reason_code` and `message`; `scope` is optional. Both arrays are empty when
there is nothing to report. ```json
{
"schema_version": "notarius.warnings.v2",
"group_count": 0,
"occurrence_count": 0,
"groups": []
}
```
It contains only process warnings. `group_count` is exact, and
`occurrence_count` is the exact sum of its group occurrence counts.
`diagnostics.json` is always the `notarius.diagnostics.v1` envelope:
```json
{
"schema_version": "notarius.diagnostics.v1",
"group_count": 0,
"occurrence_count": 0,
"truncated": false,
"unrepresented_occurrence_count": 0,
"groups": []
}
```
It contains only advisory and observation groups. `group_count` counts groups
represented in `groups`; `occurrence_count` includes both represented and
unrepresented occurrences. When `truncated` is true,
`unrepresented_occurrence_count` is the exact number omitted from group
representation.
Each group has `disposition`, `category`, `reason_code`, framework-owned
`origin`, exact `occurrence_count`, bounded `samples`, and
`omitted_sample_count`. Samples carry safe `scope` and `message`, plus a chunk
ID and zero-based chunk index when applicable. A group retains at most three
distinct samples. The framework fails rather than truncating actionable
warnings beyond 128 groups; it represents at most 256 advisory/observation
groups and records further occurrences through the diagnostic truncation
fields above.
## Compatibility ## Compatibility

View File

@@ -9,18 +9,22 @@ Command syntax, streams, and exit statuses are defined in the
## Schema ## Schema
The current schema version is `notarius.run-result.v1`. The current schema version is `notarius.run-result.v2`.
| Field | Required | Meaning | | Field | Required | Meaning |
| --- | --- | --- | | --- | --- | --- |
| `schema_version` | Yes | Exactly `notarius.run-result.v1`. | | `schema_version` | Yes | Exactly `notarius.run-result.v2`. |
| `run_id` | Yes | The finalized Notarius run identifier. | | `run_id` | Yes | The finalized Notarius run identifier. |
| `pipeline_id` | Yes | The effective pipeline identifier. | | `pipeline_id` | Yes | The effective pipeline identifier. |
| `output_directory` | Yes | Absolute path to the published, run-specific output bundle. | | `output_directory` | Yes | Absolute path to the published, run-specific output bundle. |
| `index_file` | For the production JSON output | Logical path `index.json`; omitted for other output modules. | | `index_file` | For the production JSON output | Logical path `index.json`; omitted for other output modules. |
| `normalized_output_count` | Yes | Number of final normalized outputs. | | `normalized_output_count` | Yes | Number of final normalized outputs. |
| `rejected_output_count` | Yes | Number of recorded rejected outputs. | | `rejected_output_count` | Yes | Number of recorded rejected outputs. |
| `warning_count` | Yes | Number of final run warnings. | | `warning_group_count` | Yes | Exact number of actionable warning groups. |
| `warning_occurrence_count` | Yes | Exact occurrences represented by actionable warning groups. |
| `diagnostic_group_count` | Yes | Number of represented advisory and observation groups. |
| `diagnostic_occurrence_count` | Yes | Advisory and observation occurrences, including unrepresented occurrences. |
| `diagnostics_truncated` | Yes | Whether advisory/observation group representation was truncated. |
| `validation_status` | Yes | The final run manifest validation status. | | `validation_status` | Yes | The final run manifest validation status. |
| `validation_summaries` | No | Bounded per-producer validation outcomes; present when producer work ran. | | `validation_summaries` | No | Bounded per-producer validation outcomes; present when producer work ran. |
| `debug_directory` | No | Absolute path to the run-specific debug bundle when requested debug capture completed. | | `debug_directory` | No | Absolute path to the run-specific debug bundle when requested debug capture completed. |
@@ -34,14 +38,18 @@ means one or more otherwise accepted results advanced under validator-failure
```json ```json
{ {
"schema_version": "notarius.run-result.v1", "schema_version": "notarius.run-result.v2",
"run_id": "run-1770000000000000000-0123456789abcdef0123456789abcdef", "run_id": "run-1770000000000000000-0123456789abcdef0123456789abcdef",
"pipeline_id": "dnd-session", "pipeline_id": "dnd-session",
"output_directory": "/work/results/run-1770000000000000000-0123456789abcdef0123456789abcdef", "output_directory": "/work/results/run-1770000000000000000-0123456789abcdef0123456789abcdef",
"index_file": "index.json", "index_file": "index.json",
"normalized_output_count": 6, "normalized_output_count": 6,
"rejected_output_count": 2, "rejected_output_count": 2,
"warning_count": 1, "warning_group_count": 1,
"warning_occurrence_count": 2,
"diagnostic_group_count": 3,
"diagnostic_occurrence_count": 5,
"diagnostics_truncated": false,
"validation_status": "incomplete", "validation_status": "incomplete",
"validation_summaries": [ "validation_summaries": [
{ {

View File

@@ -115,6 +115,16 @@ to checkpoint identity and `pipeline.RunInput`. The public flag and stability
contract are defined by the [CLI reference](../cli.md#run); framework and LLM contract are defined by the [CLI reference](../cli.md#run); framework and LLM
packages only transport the supplied value. packages only transport the supplied value.
The CLI also owns the scope grammar for reference flags. It enumerates the
selected chunk and lane targets from registered module specifications, expands
pipeline- and lane-scoped actions into exact stage-and-lane bindings, and
resolves overlapping bind and unbind actions by specificity before calling
configuration resolution. The generic pipeline therefore receives only exact
`ReferenceBinding` and `ReferenceUnbind` values and has no knowledge of CLI
selector syntax. Configuration resolution retains ownership of configured
external/generated conflicts, required slots, and module compatibility; file
materialization still occurs afterward.
For `run --json`, the CLI constructs and encodes its private run-result receipt For `run --json`, the CLI constructs and encodes its private run-result receipt
after a successful runner result is available, before it publishes logical after a successful runner result is available, before it publishes logical
output files. It writes the prepared receipt to standard output only after output files. It writes the prepared receipt to standard output only after
@@ -166,8 +176,9 @@ is discoverable.
- **internal/cli/production_contract_test.go** covers registrar composition, - **internal/cli/production_contract_test.go** covers registrar composition,
production catalog contents, assets, and representative configuration production catalog contents, assets, and representative configuration
validation. validation.
- **internal/cli/reference_contract_test.go** covers CLI reference overrides, - **internal/cli/reference_contract_test.go** covers scoped CLI reference
origin separation, and materialization boundaries. expansion, specificity, bind/unbind conflicts, generated-reference
protection, origin separation, and materialization boundaries.
- **internal/cli/state_hardening_test.go** covers safe run identity, state - **internal/cli/state_hardening_test.go** covers safe run identity, state
roots, and failure ordering. roots, and failure ordering.

View File

@@ -29,13 +29,17 @@ The D&D registrar registers the familys artifact codecs, extractors, typed
append-order mergers, normalizers, validators, prompt assets, fallback LLM append-order mergers, normalizers, validators, prompt assets, fallback LLM
profile asset, and default validator chains. Each extractor and normalizer has profile asset, and default validator chains. Each extractor and normalizer has
a stable module spec, explicit execution class, strict option decoding, and a a stable module spec, explicit execution class, strict option decoding, and a
typed builder. Scene chunking, every extractor, and NPC, location, and item-registry typed builder. Scene chunking, every extractor, and NPC, location, and
normalization are registered as `llm_backed`; the remaining current D&D mergers item-registry normalization are registered as `llm_backed`; the remaining
and normalizers are `deterministic`. The metadata is available to catalog inspection and current D&D mergers and normalizers are `deterministic`. The metadata is
resolved-pipeline debug data and determines which selected bindings inherit the available to catalog inspection and resolved-pipeline debug data and determines
pipeline profile. The registry normalizers use `single_response_v1`, forwarding which selected bindings inherit the pipeline profile. The registry normalizers
corrections to their reconciliation completion and retaining the accepted raw use `single_response_v1`, forwarding corrections to their reconciliation
proposal only as an owned model candidate. Configuration remains the canonical owner of the exact keys, completion and retaining the accepted raw proposal only as an owned model
candidate. When deterministic proposal safety or typed application rejects a
group, they provide separate model-facing prose for a corrective module retry;
internal issue categories, reason codes, and operator messages remain
diagnostic-only. Configuration remains the canonical owner of the exact keys,
profile precedence, and validator order. profile precedence, and validator order.
Private structured-LLM response schemas are deliberately minimal. They reject Private structured-LLM response schemas are deliberately minimal. They reject
@@ -117,14 +121,27 @@ presentation, and final ephemeral generic transcript windows. These orders and
cache controls are prompt behavior; change them only through the owning cache controls are prompt behavior; change them only through the owning
manifest and prompt declaration. manifest and prompt declaration.
NPC, item, and location registry reconciliation translate shared proposal
safety categories into bounded prose that references only the response-local
duplicate-group ordinals and candidate handles. Item reconciliation appends its
typed rule that currency may be consolidated only with aliases of the same
denomination and never with non-currency items. The next normalize attempt
receives that prose with the exact defective proposal under the shared
replacement-request protocol. Structurally invalid output has no valid proposal
candidate and receives a fresh feedback-free attempt instead. If the stage
budget is exhausted, safe groups stay applied, unsafe groups stay separate, and
one fallback warning summarizes the final defect without raw model content.
## Evidence, Candidates, And Normalization ## Evidence, Candidates, And Normalization
The current transcript is the only durable evidence source. Extractors assign The current transcript is the only durable evidence source. Extractors assign
the current source identity, preserve candidate evidence ranges for validators, the current source identity and losslessly order any reversed range whose two
and canonically order or remove exact duplicate ranges without asking the endpoints resolve in that source, using transcript position rather than numeric
model to repair semantic errors. Campaign context and generated artifacts may unit-ID order. They then canonically order ranges and remove exact duplicates.
ground names or control routing, but they never establish evidence for a D&D This routine canonicalization does not request a retry or emit a warning.
result. Unresolvable or wrong-source ranges remain unchanged for validators to reject.
Campaign context and generated artifacts may ground names or control routing,
but they never establish evidence for a D&D result.
Default chains keep responsibilities separate: structural validators assess the Default chains keep responsibilities separate: structural validators assess the
candidate, source-reference validators resolve cited ranges against the current candidate, source-reference validators resolve cited ranges against the current
@@ -134,22 +151,119 @@ relatedness validators report advisory evidence concerns. The configured order
is documented in is documented in
[Configuration](../config.md#production-validator-keys-and-default-chains). [Configuration](../config.md#production-validator-keys-and-default-chains).
Every D&D rejection describes the correction in transcript-grounded domain The optional `extract/dnd/scene-descriptions/combat_semantics` validator is the
terms, using contextual names, artifact fields, and source segment ranges when D&D family's LLM-backed review of only combat versus non-combat classification.
useful. The guidance must not ask the model to reproduce durable entity IDs, It selects the shared combat-policy prompt fragment and asks the model to
hashes, validator module keys, or reason codes. Those identifiers remain in classify the current chunk independently as `combat` or `non_combat` without
ordinary validation provenance; only the actionable semantic guidance is receiving the proposed scene kind. Deterministic code compares that
eligible for the correction prompt. classification with the proposed kind and either approves it or produces the
appropriate correction guidance. This keeps every schema-valid classification
interpretable and avoids anchoring the reviewer on the producer's answer. It
does not assess titles, summaries, non-combat subtype, or scene boundaries;
deferred boundary-coherence review remains separate. It is opt-in;
[Configuration](../config.md) owns selection and retry/failure behavior.
### Combat-semantics provider evaluation
The human-reviewed corpus at
`internal/modules/dnd/validate/scenedescriptions/combat_semantics/testdata/evaluation_cases.json`
owns the proposed kind, expected combat classification, and reviewer rationale
for each synthetic case. Its package test validates the fixture contract only.
Provider evaluation remains an explicit maintainer operation and must not be
added to the default offline test suite.
Use the following protocol before proposing default-chain inclusion:
1. Record the Notarius commit, prompt and schema fingerprints, provider, model,
profile settings, reasoning effort, structural-repair setting, number of
repetitions, and evaluation date before collecting results. Do not revise
expected classifications merely to agree with provider output; a substantive
corpus correction requires independent human review.
2. Exercise the production validator construction and prompt assets from an
explicitly invoked, disposable evaluation driver or test in the validator
package. For each corpus case, construct transcript source units from the
listed IDs and text, assign matching per-unit source references, and use
`source.MaterializeChunkPlan` with one range spanning those units. Construct
exactly one scene whose ID and source range match that chunk and whose kind
is the case's `proposed_kind`; title and summary may use fixed placeholders
because the validator neither receives nor evaluates them. Invoke the typed
validator at extract stage through the production registry and scheduled LLM
client. Do not commit provider credentials, generated source material, or an
always-on live test.
3. Compare the model classification with `expected_classification`, then verify
that its deterministic comparison with `proposed_kind` yields the expected
approval or rejection direction.
Record an unexpected approval of an expected rejection as a false
acceptance, an unexpected rejection of an expected approval as a false
rejection, and any validator execution failure separately from semantic
accuracy. Retain per-case results so repeated trials and systematic failure
modes remain visible.
4. Evaluate producer correction separately with representative complete
scene-description runs configured as shown in
[Configuration](../config.md#production-validator-keys-and-default-chains).
For every initial semantic rejection, record whether the next producer
attempt returns the requested combat status and is approved. Do not count a
PromptKit structural repair as a producer-correction attempt.
5. Run the correction evaluation with debug capture enabled and without reused
extraction checkpoints. Record added validator and producer calls, elapsed
latency, and cumulative prompt, completion, cached, and total token usage
from the debug attempt records. Compare these values with an otherwise
identical run whose scene-description chain omits the semantic validator.
The default-chain review must consider classification error, false acceptance,
false rejection, execution failure, producer-correction success, added calls,
latency, and token use together. A structurally successful provider run alone
is not evidence that the validator should become a default.
Every producer-correctable D&D rejection describes all currently detectable
corrections in transcript-grounded domain terms, using contextual names,
model-owned artifact fields, and source segment ranges when useful. Validators
collect independent record defects in one pass so one retry does not merely
reveal the next issue. Shared D&D diagnostic helpers keep repeated rules and
record descriptions stable, de-duplicated, and bounded; each artifact family
continues to own the semantic rule and its prose.
Operator diagnostics and model guidance are separate products of the same
assessment. Operator messages may use typed paths, reason details, and opaque
application identities. Correction guidance must not copy those messages or
ask the model to reproduce durable entity IDs, hashes, validator module keys,
reason codes, or Go field paths. A registry-normalization rejection instead
speaks in terms of the duplicate-group proposal response the normalizer can
actually revise. Normalization-only deterministic invariants retain useful
operator detail but do not imply that a model controls derived ordering or
identity. Only bounded actionable semantic guidance is eligible for a
correction prompt.
Private LLM schemas use simple enums for closed categorical fields when the
provider-compatible shape can express the rule directly. Deterministic typed
validators retain the same checks as defense in depth and for non-LLM
producers. Private schemas keep every property required and avoid optional
properties, `uniqueItems`, and conditional cross-field logic.
Item-occurrence shape validation groups repeated holder mistakes by occurrence
kind and gives the producer the required JSON null/non-null relationship. It
identifies affected records by contextual item name and cited transcript range,
never by the deterministically attached durable item ID. Holder mistakes remain
semantic rejections rather than silent rewrites because changing a holder can
also change the meaning of the occurrence kind.
Enemy-event extraction additionally rejects a second `engaged` observation for Enemy-event extraction additionally rejects a second `engaged` observation for
the same comparison identity within one scene-scoped result. Normalization may the same comparison identity within one scene-scoped result. Normalization may
combine results from distinct scenes, so it intentionally does not apply that combine results from distinct scenes, so it intentionally does not apply that
rule. Configuration owns the exact validator key and chain position. rule. Configuration owns the exact validator key and chain position.
Extraction source-reference validators share one full-span chunk-containment
policy. After ordinary reference validity succeeds, the policy resolves both
endpoints through document order and requires every source unit in the
inclusive range to be present in the current chunk. It does not assume numeric
unit-ID ordering, mutate input, or weaken wrong-source and unresolved-reference
validation. Scene descriptions remain separate because their validator owns an
exact one-scene range contract rather than general extraction containment.
Normalizers are deterministic for spells, combat turns, item occurrences, NPC Normalizers are deterministic for spells, combat turns, item occurrences, NPC
occurrences, scene descriptions, enemy events, and location occurrences. They occurrences, scene descriptions, enemy events, and location occurrences. They
canonicalize display values and evidence, use source-document order for stable canonicalize display values and evidence, use source-document order for stable
output, and issue bounded warnings for changes or collapsed duplicates. NPC, output, and emit bounded normalization observations for changes or collapsed duplicates. NPC,
item, and location registry normalizers are intentional exceptions: each first item, and location registry normalizers are intentional exceptions: each first
produces a deterministic candidate set, then may use a bounded structured-LLM produces a deterministic candidate set, then may use a bounded structured-LLM
proposal to reconcile identity groups. proposal to reconcile identity groups.
@@ -158,13 +272,16 @@ proposal to reconcile identity groups.
The three registry normalizers instantiate the domain-neutral The three registry normalizers instantiate the domain-neutral
`internal/framework/semanticreconcile` engine with default bounds. Each `internal/framework/semanticreconcile` engine with default bounds. Each
eligible candidate receives a contiguous, one-based `candidate_id` for that eligible candidate receives a contiguous, one-based `candidate_number` for that
request. The model sees that handle, the candidate label and source-free request, and candidate material declares the exact inclusive range. The model
evidence ranges, plus bounded transcript windows; it returns only duplicate sees that handle, the candidate label and source-free evidence ranges, plus
bounded transcript windows; it returns only duplicate
groups of supplied handles and one supplied canonical handle per group. It groups of supplied handles and one supplied canonical handle per group. It
never returns names, evidence, durable IDs, or replacement records. Identical never returns names, evidence, durable IDs, or replacement records. Identical
labels and evidence remain independently selectable because their handles are labels and evidence remain independently selectable because their handles are
distinct. distinct. Transcript unit `id` values and evidence `start_unit_id` and
`end_unit_id` values are source positions in a separate namespace and are
never valid candidate numbers.
The generic core owns the mandatory handle protocol, candidate and transcript The generic core owns the mandatory handle protocol, candidate and transcript
presentation, the private response schema, source-reference validation, presentation, the private response schema, source-reference validation,

View File

@@ -247,8 +247,12 @@ redaction boundary.
Structural repair does not replace pipeline retry behavior: a binding's Structural repair does not replace pipeline retry behavior: a binding's
configured retry count reruns its complete stage attempt after an operational configured retry count reruns its complete stage attempt after an operational
or structural error, module-requested retry, or actionable semantic rejection. or structural error, module-requested retry, or actionable semantic rejection.
The pipeline owns attempt lifecycle, validation chains, and retry diagnostics; An actionable module-requested retry and a validator rejection both use the
see [Pipeline Internals](pipeline.md#validation-retries-and-output) and the same correction payload when the producer exposes an exact latest response;
feedback-free module retries reconstruct the ordinary request without appended
messages. The pipeline owns attempt lifecycle, validation chains, and retry
diagnostics; see
[Pipeline Internals](pipeline.md#validation-retries-and-output) and the
[binding reference](../config.md#module-bindings-and-validators). [binding reference](../config.md#module-bindings-and-validators).
## Timeout Ownership ## Timeout Ownership

View File

@@ -95,19 +95,24 @@ source-backed artifact-family normalizer projects its deterministic records
into contextual candidates and owned typed record envelopes, supplies its into contextual candidates and owned typed record envelopes, supplies its
chosen prompt identity and resolved LLM profile, and constructs an engine with chosen prompt identity and resolved LLM profile, and constructs an engine with
explicit limits. The core filters invalid evidence, assigns contiguous explicit limits. The core filters invalid evidence, assigns contiguous
request-local integer handles, renders bounded candidate and transcript one-based request-local candidate numbers, renders bounded candidate and
materials, invokes the structured-completion boundary, and assesses the transcript materials, invokes the structured-completion boundary, and assesses
returned duplicate groups into a stable non-overlapping plan. the returned duplicate groups into a stable non-overlapping plan. Candidate
material declares the exact inclusive number range for the request. Transcript
unit IDs and evidence range endpoints remain source positions in a separate
namespace and are never valid candidate numbers.
The normalizer then applies that plan through a typed `ApplicationPolicy`. The The normalizer then applies that plan through a typed `ApplicationPolicy`. The
core preserves ungrouped records, contribution order, and provenance while the core preserves ungrouped records, contribution order, and provenance while the
artifact family owns group guards, field and evidence consolidation, durable artifact family owns group guards, field and evidence consolidation, durable
ID derivation, retry and fallback presentation, warnings, and postconditions. ID derivation, retry and fallback presentation, classified diagnostics, and
Request-local handles do not enter the typed value or durable artifact. Fewer postconditions.
than two eligible candidates skips model invocation; exceeding a candidate or These candidate numbers are the concrete private representation of ADR-0013's
combined-material bound preserves the deterministic result under the family's request-local handles; they do not enter the typed value or durable artifact.
fallback policy. Provider, transport, cancellation, and context-construction Fewer than two eligible candidates skips model invocation; exceeding a
failures remain execution errors. candidate or combined-material bound preserves the deterministic result under
the family's fallback policy. Provider, transport, cancellation, and
context-construction failures remain execution errors.
When the engine actually makes a proposal call, its typed result carries the When the engine actually makes a proposal call, its typed result carries the
owned exact proposal response under the same correction contract as other owned exact proposal response under the same correction contract as other
@@ -115,6 +120,16 @@ eligible producers. Deterministic skip, limit, and fallback outcomes carry no
model candidate, so a later rejection applies terminal policy without spending model candidate, so a later rejection applies terminal policy without spending
an ineffective semantic retry. an ineffective semantic retry.
When proposal assessment or typed group application rejects a structurally
valid group, the normalizer may return its safe partial value with a
module-requested retry. A feedback-capable directive supplies bounded
model-facing correction guidance separately from operator diagnostics and
retains the exact proposal response as its candidate. The shared stage retry
mechanism appends that response and guidance to a fresh complete request. A
structurally invalid completion has no valid candidate and therefore requests a
feedback-free fresh attempt. On exhaustion, only the final safe fallback and
its bounded process diagnostic advance to validation.
The core supplies a conservative generic prompt and the single private The core supplies a conservative generic prompt and the single private
response schema. A domain prompt may substitute its semantic instructions but response schema. A domain prompt may substitute its semantic instructions but
mounts the core-owned protocol and candidate/transcript presentation assets. mounts the core-owned protocol and candidate/transcript presentation assets.

View File

@@ -12,7 +12,7 @@ own durable output shapes. Concrete production extensions are covered by
The pipeline framework accepts a resolved composition, registries, shared The pipeline framework accepts a resolved composition, registries, shared
dependencies, input bytes, a supplied prompt session, and state/debug dependencies, input bytes, a supplied prompt session, and state/debug
collaborators. It returns logical output files, normalized artifacts, recorded collaborators. It returns logical output files, normalized artifacts, recorded
rejections and warnings, manifest provenance, and checkpoint decisions. The rejections, grouped diagnostics, manifest provenance, and checkpoint decisions. The
CLI owns process arguments, configuration discovery, session resolution, CLI owns process arguments, configuration discovery, session resolution,
physical roots, and placement of returned output files. physical roots, and placement of returned output files.
@@ -97,6 +97,13 @@ codec, checks its complete schema and media identity, and records a content
digest plus bounded producer provenance. A missing, ambiguous, invalid, or digest plus bounded producer provenance. A missing, ambiguous, invalid, or
incompatible producer prevents the consumer step from starting. incompatible producer prevents the consumer step from starting.
Resolution receives only exact stage-and-lane operational reference overrides.
The CLI may offer broader pipeline- or lane-scoped selectors, but expands and
arbitrates those before entering the framework. External overrides are applied
after configured external defaults and local bindings. They cannot coexist
with a generated binding for the same target and slot, and external unbinds do
not remove generated handoffs.
## Execution And Ordering ## Execution And Ordering
The runner validates its input, installs no-op state collaborators when none The runner validates its input, installs no-op state collaborators when none
@@ -150,34 +157,41 @@ oversized aggregate is a framework contract error; guidance is never inferred
or truncated. or truncated.
The runner applies the binding's retry policy around a stage operation and its The runner applies the binding's retry policy around a stage operation and its
complete validation chain. It preserves warnings only from the final accepted complete validation chain. It preserves terminal diagnostics only from the final accepted
or rejected attempt, plus one fixed warning per validator whose execution or rejected attempt, plus one fixed validation-incomplete warning per validator whose execution
budget was exhausted under `warn_continue`. Cancellation stops retries. budget was exhausted under `warn_continue`. Cancellation stops retries.
Normalizer-specific retry directives consume this same budget and validate any Normalizer-specific retry directives consume this same budget and validate any
final safe fallback through the normalizer chain. final safe fallback through the normalizer chain. A directive may carry bounded
correction guidance only when it also exposes the exact latest
`single_response_v1` candidate. The state machine then uses the same replacement
request shape as validator correction. A directive without guidance clears any
prior correction and starts a fresh attempt, which preserves structural retry
behavior when no valid response exists.
The artifact-neutral producer-attempt state machine owns that shared budget, The artifact-neutral producer-attempt state machine owns that shared budget,
attempt provenance, semantic-correction material, and terminal-policy attempt provenance, semantic-correction material, and terminal-policy
selection. It accepts producer and complete-validation closures, so artifact selection. It accepts producer and complete-validation closures, so artifact
materialization, cache handling, checkpoints, and debug output stay at the materialization, cache handling, checkpoints, and debug output stay at the
operation boundary. It distinguishes operational, structural, module-requested, operation boundary. It distinguishes operational, structural, module-requested,
and semantic retries. A semantic retry is available only for a valid latest and validator-semantic retries. Model feedback from either semantic source is
`single_response_v1` candidate; a deterministic or no-model rejection instead available only for a valid latest `single_response_v1` candidate. A
settles the semantic policy immediately. Structural-output errors alone use the deterministic or no-model validator rejection instead settles the semantic
structural policy, and validation failure without rejection settles the policy immediately, while a feedback-free module directive remains an ordinary
validator-failure policy without regenerating the producer. fresh retry. Structural-output errors alone use the structural policy, and
validation failure without rejection settles the validator-failure policy
without regenerating the producer.
Chunk planning uses this state machine for generated plans. A rejected or Chunk planning uses this state machine for generated plans. A rejected or
validation-incomplete automatic cache hit is not model material and therefore validation-incomplete automatic cache hit is not model material and therefore
falls through to a fresh initial generation at producer attempt one; it neither falls through to a fresh initial generation at producer attempt one; it neither
receives a correction, consumes retry budget, promotes cached-candidate receives a correction, consumes retry budget, promotes cached-candidate
warnings, nor overwrites the stored record. An incomplete cache validation diagnostics, nor overwrites the stored record. An incomplete cache validation
under `fail_run` terminates instead. Only a newly generated, completely under `fail_run` terminates instead. Only a newly generated, completely
validated plan is published to the chunk-plan store. Rejected plans never validated plan is published to the chunk-plan store. Rejected plans never
advance, and validation-incomplete plans remain unpublishable. advance, and validation-incomplete plans remain unpublishable.
After terminal lane work, the runner assembles manifest provenance, normalized After terminal lane work, the runner assembles manifest provenance, normalized
artifacts, rejections, warnings, and an optional accepted chunk map. When an artifacts, rejections, final grouped diagnostics, and an optional accepted chunk map. When an
output policy selected evidence lanes, it decodes accepted serialized normalize output policy selected evidence lanes, it decodes accepted serialized normalize
outputs through their registered codecs and invokes the prepared typed outputs through their registered codecs and invokes the prepared typed
projectors. Rejected or absent lanes contribute nothing. This reconstruction is projectors. Rejected or absent lanes contribute nothing. This reconstruction is
@@ -224,6 +238,8 @@ Debug recording is attempt-scoped and application-owned. A failure to persist
required debug data is a framework error. State roots, persistence, reason-code required debug data is a framework error. State roots, persistence, reason-code
meanings, resume, and cleanup are intentionally owned by meanings, resume, and cleanup are intentionally owned by
[Run State Internals](state.md) and [Operations](../operations.md). [Run State Internals](state.md) and [Operations](../operations.md).
Extract-validator trace scopes include the current chunk ordinal so concurrent
chunks cannot overwrite one another's validator attempts or LLM artifacts.
## Invariants To Preserve ## Invariants To Preserve

View File

@@ -64,12 +64,13 @@ Ordinary resume loads extract, merge, and normalize checkpoints progressively
and may execute later lane stages after an earlier cache miss. Selective and may execute later lane stages after an earlier cache miss. Selective
recomputation instead asks the loader for the required producer's accepted recomputation instead asks the loader for the required producer's accepted
normalize artifact. That lookup reuses the existing normalize files, requires normalize artifact. That lookup reuses the existing normalize files, requires
workspace schema v3 plus an exact non-empty invocation identity, and deliberately workspace schema v4 plus an exact non-empty invocation identity, and deliberately
does not require extract or merge checkpoint files or dependency fingerprints. does not require extract or merge checkpoint files or dependency fingerprints.
The runner performs canonical codec and producer-provenance validation before The runner performs canonical codec and producer-provenance validation before
cloning the artifact into normal step output. Success restores only stored cloning the artifact into normal step output. Success restores only stored
normalize warnings and emits one normalize decision; failure retains the files, normalize diagnostics and emits one normalize decision; failure retains the
records the decision, and stops without executing the producer or consumer. files, records the decision, and stops without executing the producer or
consumer.
The loader assigns a typed category and reason code at each validation site; The loader assigns a typed category and reason code at each validation site;
diagnostic prose is not classified after the fact. The runner then applies diagnostic prose is not classified after the fact. The runner then applies
@@ -97,7 +98,7 @@ owns the operator workflow and stable reason-code meanings.
`internal/core/debugbundle` allocates an explicitly requested per-run bundle `internal/core/debugbundle` allocates an explicitly requested per-run bundle
with `summary/` and `trace/` roots. `SummaryWriter` persists redacted command, with `summary/` and `trace/` roots. `SummaryWriter` persists redacted command,
resolution, run, warning, and failure artifacts. `internal/framework/debug` resolution, run, final grouped diagnostic, and failure artifacts. `internal/framework/debug`
implements the pipeline-facing trace recorder under the trace root. implements the pipeline-facing trace recorder under the trace root.
The CLI allocates a bundle before pipeline resolution and treats requested The CLI allocates a bundle before pipeline resolution and treats requested

View File

@@ -105,8 +105,10 @@ and resolves configuration before module preparation and source parsing. It
then performs any permitted cache lookup, executes the pipeline, and publishes then performs any permitted cache lookup, executes the pipeline, and publishes
logical output files only after a successful runner result. logical output files only after a successful runner result.
On success, the command reports the output bundle path. A warning-bearing run On success, the command reports the output bundle path. A run with actionable
still succeeds and reports its warning count on standard error. Errors and process warnings still succeeds and reports warning-group and occurrence counts
on standard error; advisory and observation findings do not produce a warning
line. Errors and
their exit classes are defined in the [CLI reference](cli.md#output-streams-and-exit-statuses). their exit classes are defined in the [CLI reference](cli.md#output-streams-and-exit-statuses).
## Validation Retries And Terminal Outcomes ## Validation Retries And Terminal Outcomes
@@ -127,13 +129,14 @@ validator execution failure normally uses `warn_continue`, which keeps an
otherwise accepted result in the current run with `incomplete` validation otherwise accepted result in the current run with `incomplete` validation
provenance. It emits one bounded warning for every validator whose execution provenance. It emits one bounded warning for every validator whose execution
budget was exhausted. A corrected result that later passes validation does not budget was exhausted. A corrected result that later passes validation does not
retain abandoned-attempt warnings. retain diagnostics from abandoned attempts.
Treat a successful process exit as a completed run, not as proof that every Treat a successful process exit as a completed run, not as proof that every
candidate was fully validated. Inspect the receipt's `validation_status`, candidate was fully validated. Inspect the receipt's `validation_status`,
`validation_summaries`, rejection count, and warning count when an orchestrator `validation_summaries`, rejection count, and warning group and occurrence
requires complete validation. The durable fields and their meanings are owned counts when an orchestrator requires complete validation. The durable fields
by the [run-result receipt](integrations/run-result.md) and and their meanings are owned by the
[run-result receipt](integrations/run-result.md) and
[published JSON output contract](integrations/json-output.md). [published JSON output contract](integrations/json-output.md).
## Output Bundles ## Output Bundles
@@ -272,7 +275,7 @@ Only a [debug-enabled run](cli.md#run) creates a bundle:
~~~ ~~~
The summary contains redacted invocation and resolution information plus run, The summary contains redacted invocation and resolution information plus run,
warning, checkpoint, chunk-plan, and terminal reporting artifacts. Attempt final grouped diagnostic, checkpoint, chunk-plan, and terminal reporting artifacts. Attempt
terminal records contain bounded attempt kinds, validator outcomes, policy, terminal records contain bounded attempt kinds, validator outcomes, policy,
decision, PromptKit repair count, and usage; they do not contain assistant decision, PromptKit repair count, and usage; they do not contain assistant
responses or complete correction messages. The trace contains allowlisted responses or complete correction messages. The trace contains allowlisted

View File

@@ -146,8 +146,9 @@ lanes, validators, and LLM profile: the canonical source digest selects the
plan, while the current run still applies its configured chunk validators to plan, while the current run still applies its configured chunk validators to
the materialized chunks. the materialized chunks.
The framework owns orchestration and handoff provenance. Modules return logical The framework owns orchestration, origin enrichment, aggregation, and handoff
results and warnings; they do not own CLI reporting, physical output, cache, or provenance. Modules return logical results and classified diagnostics; they do
not own CLI reporting, physical output, cache, or
debug roots, durable file placement, or checkpoint and debug lifecycle. debug roots, durable file placement, or checkpoint and debug lifecycle.
After pipeline-wide chunking, extraction uses bounded framework concurrency. After pipeline-wide chunking, extraction uses bounded framework concurrency.
@@ -159,17 +160,26 @@ may overlap. The framework must not create unbounded goroutines per lane or
chunk. chunk.
Completion timing does not choose public ordering or errors. The coordinator Completion timing does not choose public ordering or errors. The coordinator
orders accepted artifacts, warnings, rejections, checkpoint events, and orders accepted artifacts, grouped diagnostics, rejections, checkpoint events, and
framework errors by stable pipeline scope. Rejections do not cancel unrelated framework errors by stable pipeline scope. Rejections do not cancel unrelated
work. A framework error cancels derived work, prevents undispatched work from work. A framework error cancels derived work, prevents undispatched work from
starting, waits for started work, and prevents output encoding. starting, waits for started work, and prevents output encoding.
Warnings are process-only signals: configuration degradation, approved fallback,
or incomplete configured validation. Quality uncertainty and grounding findings
are advisories; successful canonicalization and cleanup are observations.
Modules choose that semantic classification, while the framework attaches
origin, aggregates groups, enforces bounds, and presents final collections.
An ordinary successful run therefore has zero warnings. See
[ADR-0015](../adr/0015-separate-process-warnings-from-quality-diagnostics.md)
for the decision rationale.
## Validation ## Validation
Validation is a framework-managed boundary around outputs from chunk, extract, Validation is a framework-managed boundary around outputs from chunk, extract,
merge, and normalize stages. Validators receive immutable stage output and merge, and normalize stages. Validators receive immutable stage output and
make an explicit whole-output decision: approve, approve with warnings, make an explicit whole-output decision: approve, reject, fail, or skip when a
reject, fail, or skip when a runtime prerequisite is unavailable. runtime prerequisite is unavailable.
Typed artifact validators receive the domain value directly. Chunk validators Typed artifact validators receive the domain value directly. Chunk validators
receive source-zone chunks, while serialized validators receive immutable receive source-zone chunks, while serialized validators receive immutable
@@ -191,6 +201,13 @@ they are not model instructions. The framework constructs model-facing retry
text only from the semantic guidance and fails the contract rather than text only from the semantic guidance and fails the contract rather than
inventing or truncating missing guidance. inventing or truncating missing guidance.
An LLM-backed module may also request a feedback-aware retry when its own
deterministic translation or typed safety policy rejects a structurally valid
model response. It must supply model-facing guidance separately from its
reason code, operator message, and fallback diagnostics, together with the
exact `single_response_v1` candidate. A feedback-free module retry remains
valid when no exact candidate exists.
Default validator chains are production composition policy and are registered Default validator chains are production composition policy and are registered
centrally by stage and module. Configuration may replace a stage-local default, centrally by stage and module. Configuration may replace a stage-local default,
including with an explicitly empty chain. Configured validator order is including with an explicitly empty chain. Configured validator order is
@@ -212,12 +229,14 @@ the two budgets must remain separate.
An LLM-backed producer can participate in semantic correction only when it An LLM-backed producer can participate in semantic correction only when it
declares `single_response_v1` and returns the exact one response that directly declares `single_response_v1` and returns the exact one response that directly
controlled its candidate. On an actionable rejection, the framework rebuilds controlled its candidate. On an actionable validator rejection or
the ordinary request and appends only the latest defective response as an feedback-capable module retry, the framework rebuilds the ordinary request and
`assistant` message plus one aggregated `user` correction message. This is a appends only the latest defective response as an `assistant` message plus one
fresh replacement request, not a growing conversation. The retry budgets, aggregated `user` correction message. This is a fresh replacement request, not
terminal policy, and sensitive-data rationale are recorded in a growing conversation. The retry budgets, terminal policy, and sensitive-data
[ADR-0014](../adr/0014-feedback-aware-validation-retries.md). rationale are recorded in
[ADR-0014](../adr/0014-feedback-aware-validation-retries.md) and
[ADR-0016](../adr/0016-feedback-aware-module-requested-retries.md).
When a model selects an application entity, callers must supply a contextual When a model selects an application entity, callers must supply a contextual
selection and deterministically attach the opaque application identity whenever selection and deterministically attach the opaque application identity whenever

85
docs/releases/v0.5.0.md Normal file
View File

@@ -0,0 +1,85 @@
# Notarius v0.5.0
This release separates actionable process warnings from extraction-quality
advisories and routine normalization observations, giving operators a quiet
warning channel without discarding durable diagnostic detail.
## Summary
Notarius now carries one validated, origin-aware diagnostic contract from
producers and validators through retries, reusable state, output publication,
debug summaries, run receipts, and CLI presentation. Warnings are reserved for
process degradation or incomplete configured work. Data-quality findings are
advisories, and successful deterministic cleanup is recorded as observations.
An ordinary successful run therefore reports zero warnings while retaining
bounded diagnostic provenance for later review.
The framework aggregates findings deterministically by their stable identity
and complete pipeline origin, preserves exact occurrence counts, and retains
bounded representative samples. Warning groups fail rather than truncate;
advisory and observation representation is bounded with explicit truncation
metadata and exact unrepresented-occurrence counts.
## Compatibility
- `warnings.json` now uses the incompatible grouped
`notarius.warnings.v2` envelope and contains process warnings only. Consumers
of the former flat warning payload must migrate to the current
[JSON output contract](../integrations/json-output.md).
- The new `diagnostics.json` file uses `notarius.diagnostics.v1` and contains
advisory and observation groups. Production `index.json` files always expose
both `warnings_file` and `diagnostics_file`.
- The machine-readable run receipt is now `notarius.run-result.v2`. It replaces
`warning_count` with exact warning group and occurrence counts and adds
advisory/observation group, occurrence, and truncation fields. See the
current [run-result receipt](../integrations/run-result.md).
- Custom output modules must return their complete logical file set or an
error. The former `OutputResult.Warnings` field has been removed; an output
module cannot report a warning after serializing its output.
- Reusable state now uses `notarius.workspace.v4` and chunk-plan records use
`notarius.chunk-plan.v3` so they can preserve structured diagnostics. Older
pre-release reusable state is not reused under these contracts; start with
clean state when deterministic continuity with an older workspace is not
required.
- Validation acceptance, semantic retry budgets, rejection policy, and D&D
artifact schema identities are unchanged by this release.
## Upgrade
1. Update subprocess consumers to require `notarius.run-result.v2` and read
`warning_group_count`, `warning_occurrence_count`,
`diagnostic_group_count`, `diagnostic_occurrence_count`, and
`diagnostics_truncated`.
2. Update output-bundle consumers to decode `notarius.warnings.v2`, discover
`diagnostics.json` through `index.json`, and treat diagnostics as review
information rather than process warnings.
3. Update any custom output module for the removal of
`OutputResult.Warnings`; return an error when encoding cannot complete.
4. Clear pre-release reusable state before the first upgraded production run
when deterministic continuity with an older workspace is not required.
5. Run `notarius config validate --config <path> --pipeline <id>` and perform
one representative run before promoting the release in an automated
pipeline.
## Changes
- Added validated diagnostic dispositions, categories, origins, stable reason
codes, exact occurrence counts, and bounded representative samples.
- Added deterministic run-level aggregation with separate limits for
actionable warning groups and advisory/observation groups.
- Reclassified D&D source-relatedness and unresolved-identity findings as
data-quality advisories and routine normalization changes as observations.
- Preserved structured diagnostics across producer retries, validation,
generated-reference handoff, checkpoints, chunk-plan reuse, and debug
summaries while discarding superseded-attempt findings.
- Added grouped `warnings.json`, a new grouped `diagnostics.json`, and the
corresponding production index entries.
- Upgraded the machine-readable run receipt and human CLI summary to report
exact warning and diagnostic counts without allowing advisory volume to
create warning output.
- Removed post-encoding output warnings and hardened diagnostic validation,
overflow handling, aggregate memory bounds, and warning-file path
presentation.
- Documented diagnostic ownership, classification, operator interpretation,
durable contracts, and architectural invariants in ADR-0015 and the
canonical CLI, operations, integration, and internal documentation.

84
docs/releases/v0.6.0.md Normal file
View File

@@ -0,0 +1,84 @@
# Notarius v0.6.0
This release improves the reliability and ergonomics of unattended,
subprocess-driven D&D extraction pipelines.
## Summary
Notarius now gives models more precise, semantically useful correction guidance
when deterministic or model-backed validation rejects an otherwise structured
candidate. Semantic reconciliation retries identify candidates with compact,
request-local numbers, preserve valid candidates when a proposal cannot be
repaired, and report bounded process warnings when reconciliation is
incomplete. D&D extraction also canonicalizes safely resolvable reversed source
ranges and applies more consistent schema, prompt, and validator constraints,
improving successful operation with smaller models without weakening evidence
validation.
An optional LLM-backed combat-semantics validator can review whether scene
descriptions correctly consolidate combat and apply the `combat` kind. For
subprocess callers, command-line references now follow the pipeline-scoped
configuration model: a shared reference can be supplied once and automatically
reach every compatible selected target, while lane- and binding-specific forms
remain available for exceptions.
## Compatibility
- The meaning of an unqualified `--reference slot=path` or
`--without-reference slot` selector has changed. It now applies to every
selected pipeline target that declares the slot instead of requiring exactly
one matching target. Review callers that relied on ambiguity rejection or a
unique implicit target; the current selector contract is documented in the
[CLI reference](../cli.md).
- The stage-wide `merge.slot` CLI reference shorthand has been removed. Use
`lane.merge.slot` for an exact merge binding, `lane.slot` for all compatible
bindings in one lane, or an unqualified `slot` for pipeline scope.
- Command-line file references cannot replace, remove, or coexist with a
generated artifact handoff for the same concrete target and slot. Resolution
reports the conflict so the caller can narrow or remove the selector.
- The private LLM-facing semantic-reconciliation contract now uses contiguous
`candidate_number` values rather than application identities. This is not a
durable or operator-configurable contract and requires no operator action.
- Durable D&D artifact schemas, the `notarius.run-result.v2` receipt,
`notarius.warnings.v2`, and `notarius.diagnostics.v1` are unchanged from
`v0.5.0`.
## Upgrade
1. Update subprocess commands to provide shared inputs once with
`--reference slot=path`, and convert any stage-wide `merge.slot` selector to
an explicit supported scope. The [subprocess consumer guide](../consumers/subprocess.md)
and [complete D&D consumer guide](../consumers/dnd-pipeline.md) show the
current invocation pattern.
2. Review CLI reference overrides that overlap generated same-run references;
narrow or remove an external selector rather than attempting to replace the
generated handoff.
3. Run `notarius config validate --config <path> --pipeline <id>` for every
deployed pipeline configuration.
4. Optionally add
`extract/dnd/scene-descriptions/combat_semantics` to the scene-description
validator chain where the deployment wants LLM-backed combat-scene review;
see the [configuration reference](../config.md).
5. Perform a representative run with the deployed model profile and inspect
the machine-readable receipt, warnings, and diagnostics before promotion.
## Changes
- Added feedback-aware, module-requested retry support for normalizers while
keeping retry budgets bounded and preserving attempt diagnostics.
- Reworked semantic reconciliation around compact, contiguous candidate
numbers, explicit namespace guidance, complete proposal validation, and
model-facing retry feedback that omits opaque internal identities.
- Preserved candidates separately when an exhausted semantic proposal cannot
be safely applied, with a bounded process warning describing the fallback.
- Canonicalized safely resolvable reversed D&D source-reference endpoints
before deterministic coverage validation and clarified the shared evidence
prompt.
- Improved D&D schemas, prompts, and validators with closed-value constraints,
contextual correction guidance, shared source-range coverage logic, and
clearer item-holder transition rules.
- Added the optional D&D scene-description combat-semantics validator, shared
combat policy assets, evaluation fixtures, and retry-path coverage.
- Made unqualified CLI references pipeline-scoped, added hierarchical lane and
exact-binding selectors, defined deterministic override precedence, and
documented first-class subprocess use.

View File

@@ -0,0 +1,648 @@
# Warning Signal And Presentation Audit
## Executive Assessment
Notarius warning execution is mechanically stronger than its operator-facing
presentation. Terminal-attempt promotion, stable ordering after concurrent
work, checkpoint replay, validation summaries, and debug retention are all
substantially correct. The audit found no general duplicate-append defect in
the extract, merge, or normalize handoffs and no leakage of abandoned-attempt
warnings into a successful result.
The warning channel itself is not coherent. One flat `contracts.Warning` type
currently represents at least four materially different concepts:
- actionable degradation or incomplete validation;
- heuristic data-quality doubt;
- successful but potentially reviewable fallback; and
- routine canonicalization, ordering, and deduplication observations.
That conflation is the primary reason successful runs produce a count that is
large but operationally weak. The maintained complete D&D example demonstrates
the problem without a live provider: an approved run with no rejected outputs
published 12 warning records, all from three advisory relatedness checks. An
operator separately reported a successful complete D&D run with 10 outputs,
one rejection, and 85 warnings. The production bundle for that run was not
available in this environment, so its reason-code distribution could not be
measured.
The current implementation also has four correctness or robustness gaps:
1. warning records lose stage, step, lane, module, validator, and chunk
provenance when promoted, which makes safe aggregation and diagnosis
impossible from `warnings.json` alone;
2. there is no framework-level validation or aggregate bound, and the NPC- and
spell-relatedness validators bypass the D&D warning limiter entirely;
3. warnings returned by an output encoder are added after `warnings.json` has
already been encoded, so the receipt, stderr, debug bundle, and published
warning file can disagree; and
4. a skipped validator contributes to `incomplete` validation but does not
receive the warning generated for an exhausted validator failure.
The recommended end state is a structured diagnostic contract with explicit
disposition, category, origin, occurrence count, and bounded samples. Warnings
are reserved for process-level degradation or incompleteness. LLM-judged or
deterministically inferred extraction-quality signals are advisories, never
warnings, and routine normalization observations remain inspectable without
being reported as top-level warnings. An ordinary successful run in which all
configured work completes normally should therefore report zero warnings. This
is an architectural and durable-contract change, not merely revised CLI prose.
## Evidence And Limits
The audit used:
- a complete static search of production `contracts.Warning` constructors,
reason-code constants, result fields, and promotion sites under `internal/`;
- call-path inspection through producer attempts, validators, lane
coordination, chunk-plan reuse, checkpoints, output encoding, debug output,
CLI presentation, and run-result construction;
- the maintained complete and minimal D&D examples with offline fake LLMs;
- focused deterministic tests for warning bounds, semantic-reconciliation
fallback, `warn_continue`, semantic retries, terminal rejection, concurrency
ordering, and checkpoint reuse; and
- the operator-provided observation of an 85-warning complete D&D run.
No provider-backed production run was attempted because this environment has
no API key. Consequently, the audit can establish warning mechanics, possible
multiplicity, synthetic volume, and obvious heuristic limitations, but cannot
estimate production frequency or the real false-positive rate of individual
D&D advisories. Those measurements are not required to choose the recommended
architecture; they are required before strengthening any heuristic advisory
into a rejection or setting a numerical production acceptance target.
## Complete Warning-Producer Inventory
### Framework And Generic Boundaries
| Producer | Reason code | Trigger and consequence | Multiplicity and bound | Current surfaces and coverage |
| --- | --- | --- | --- | --- |
| Reference materialization in `internal/framework/pipeline/references.go` | `empty_reference` | A bound external reference is a valid, accepted media type but contains zero bytes. The prompt may receive materially incomplete context. | One per empty bound file; finite by configuration but no shared run-level cap. | Enters `RunInput.Warnings`; reference tests protect contextual scope. |
| Producer-attempt policy in `internal/framework/pipeline/producer_attempts.go` | `validator_execution_incomplete` | An applicable validator exhausted its execution budget and `warn_continue` accepted the otherwise valid candidate. | One per failed validator on each terminal candidate. An extract chain can multiply this by chunks and lanes. There is no global cap. | Durable warning, receipt count, stderr, debug, and validation summary. `TestWarnContinueRecordsOneWarningForEachExhaustedValidator` covers failures. |
| Chunk, extract, merge, normalize, and output module result contracts | Module-defined | A module may return arbitrary warnings with its successful candidate. | No contract validation, message limit, per-result cap, or global cap. Current production modules are inventoried below. | Terminal-attempt filtering and concurrency ordering are well tested. |
| Production JSON output encoder | None | The encoder copies incoming warnings into `warnings.json`; it does not currently create warnings. | Same count as its input. | JSON encoder and assembled-pipeline tests compare the incoming run warnings with the published file. |
| Output encoder result contract | Module-defined | Any output encoder may return warnings discovered during encoding. | Unbounded by contract. No production encoder currently exercises this capability. | Appended to final `RunOutput` only after logical files were encoded; this is the cross-surface defect described in AUD-WARN-004. |
Input adapters and production mergers do not currently have independent
warning producers. Chunk-plan and checkpoint decisions are structured manifest
or debug provenance rather than warnings. Cancellation and hard persistence,
reference, parsing, serialization, and provider failures remain errors.
### D&D Extraction Gates
| Producer | Reason code | Trigger and consequence | Multiplicity and bound |
| --- | --- | --- | --- |
| `dnd/combat-turns` extractor | `scene_classification_unavailable` | The chunk has no exact matching scene-description classification. The extractor returns an empty accepted result and skips the LLM, so combat-turn output may be incomplete. | At most one per chunk for this lane. |
| `dnd/enemy-events` extractor | `scene_classification_unavailable` | The same missing or mismatched scene gate causes accepted empty enemy-event output. | At most one per chunk for this lane. |
An exact non-combat classification produces an intentional empty result without
a warning. An exact combat classification proceeds normally. The two producers
share a code and operator consequence but use different messages; their module
origins are not retained in the final warning record.
### D&D Source-Relatedness Validators
All ten relatedness validators are deterministic advisories: they approve the
candidate and warn when contextual prose or an entity name is not lexically
present in cited text. Shape and source-reference failures are deliberately
left to blocking validators earlier in the chain. The same relatedness
validator is registered in both the extract and normalize default chain for
each artifact family in `internal/modules/dnd/register/chains.go`.
| Artifact family | Warning reason | Per-record trigger | Local bound | Omission reason |
| --- | --- | --- | --- | --- |
| Combat turns | `combat_turn_not_near_source` | Actor token sequence absent | 20 per validator invocation | `combat_turn_relatedness_warnings_omitted` |
| Enemy events | `enemy_event_not_near_source` | Subject token sequence absent | 20 | `enemy_event_relatedness_warnings_omitted` |
| Item occurrences | `item_occurrence_source_unrelated` | Item name token sequence absent | 20 | `item_occurrence_relatedness_warnings_omitted` |
| Item registry | `item_not_near_source` | Item name token sequence absent | 20 | `item_relatedness_warnings_omitted` |
| Location occurrences | `location_occurrence_not_near_source` | Location name token sequence absent | 20 | `location_occurrence_relatedness_warnings_omitted` |
| Location registry | `location_not_near_source` | Location name token sequence absent | 20 | `location_relatedness_warnings_omitted` |
| NPC occurrences | `npc_occurrence_not_near_source` | NPC name token sequence absent | 20 | `npc_occurrence_relatedness_warnings_omitted` |
| NPC registry | `npc_not_near_source` | NPC name token sequence absent | **Unbounded** | None |
| Scene descriptions | `scene_description_not_near_source` | No significant title or summary token appears; up to two findings per scene | 20 | `scene_description_relatedness_warnings_omitted` |
| Spells | `spell_not_near_source` | Spell-name token sequence absent | **Unbounded** | None |
The eight limiter-generated omission records are presentation artifacts, not
new source-relatedness conditions. They occupy a warning slot and make list
length differ from the actual occurrence count.
### D&D Normalizers
Every production D&D normalizer bounds its returned warning slice to 20 through
`internal/modules/dnd/shared/diagnostics`, including a final omission record
when needed. Registry semantic retries reserve space for their fallback
warning. The following table is complete by semantically distinct condition;
codes listed together are parallel artifact-family variants.
| Condition | Reason codes | Result impact | Current classification assessment |
| --- | --- | --- | --- |
| Display or field whitespace/name canonicalization | `npc_fields_normalized`, `item_fields_normalized`, `location_fields_normalized`, `spell_name_canonicalized`, `combat_actor_canonicalized`, `enemy_event_name_canonicalized`, `item_occurrence_name_canonicalized`, `location_occurrence_name_canonicalized`, `scene_description_prose_normalized` | Deterministic successful mutation. The item-occurrence code can also describe `from`/`to` whitespace, not only the item name. | Routine observation. |
| Durable ID recomputation | `npc_id_recomputed`, `item_id_recomputed`, `location_id_recomputed` | Restores the deterministic name-derived ID. | Routine observation; invalid identity is separately rejected by default chains. |
| Source-reference sorting or deduplication | `source_references_normalized` | Sorts and removes exact duplicate references while deliberately preserving invalid references for their validators. | Routine observation. Shared code is useful but ambiguous without producer origin. |
| Canonical record ordering | `combat_turns_reordered`, `enemy_events_reordered`, `item_occurrences_reordered`, `location_occurrences_reordered`, `npc_occurrences_reordered`, `scene_description_order_normalized` | Deterministic order changes only. | Routine observation. |
| Exact or approved semantic duplicate consolidation | `duplicate_npc_collapsed`, `duplicate_item_collapsed`, `duplicate_location_collapsed`, `duplicate_spell_cast_collapsed`, `duplicate_combat_turn_collapsed`, `duplicate_enemy_event_collapsed`, `duplicate_item_occurrence_collapsed`, `duplicate_location_occurrence_collapsed`, `duplicate_npc_occurrence_collapsed`, `scene_description_duplicate_collapsed` | Removes duplicate records and preserves or combines canonical evidence according to the artifact policy. Registry codes cover both exact and accepted semantic consolidation. | Durable normalization observation; not normally operator-actionable. |
| Unresolved external membership | `spell_name_unresolved`, `item_occurrence_unknown_item_id`, `location_occurrence_unknown_location_id` | The value is preserved but is not grounded in the effective catalog or registry. Default chains normally reject the same condition before normalization; it remains reachable with validator overrides or defensive direct use. | Actionable data-quality warning. |
| Unsafe currency consolidation proposal | `item_semantic_proposal_invalid` | The proposed group is rejected and all records are preserved because denominations or currency/non-currency members are incompatible. The same code is also used internally as a retry reason. | Advisory about model proposal quality; no accepted-data loss. The control and diagnostic meanings should be separated. |
| Semantic reconciliation unavailable or exhausted | `npc_semantic_reconciliation_exhausted`, `item_semantic_reconciliation_exhausted`, `location_semantic_reconciliation_exhausted` | The safe deterministic result is accepted, but possible semantic duplicates remain. | Actionable fallback warning. |
| Local warning truncation | `npc_normalization_warnings_omitted`, `item_normalization_warnings_omitted`, `location_normalization_warnings_omitted`, `spell_normalization_warnings_omitted`, `combat_turn_normalization_warnings_omitted`, `enemy_event_normalization_warnings_omitted`, `item_occurrence_normalization_warnings_omitted`, `location_occurrence_normalization_warnings_omitted`, `npc_occurrence_normalization_warnings_omitted`, `scene_description_normalization_warnings_omitted` | Reports that individual records were omitted from presentation. | Group metadata, not an independent warning. |
No production D&D normalization warning exposes raw model responses,
correction guidance, or provider errors. Most dynamic names are quoted and
truncated by the shared helper. That local discipline is not enforced by the
generic warning contract, and the spell relatedness message does not use the
shared truncation helper.
## Warning Propagation And Surface Map
```text
external-reference warnings -----------------------------+
|
module candidate warnings -> validation chain warnings |
| | |
+---- producer-attempt terminal policy -------+
| |
accepted / terminal rejection only |
| |
chunk or lane result in canonical order |
| |
checkpoint record/replay and ordered step merge |
| |
RunOutput.Warnings <------------+
|
OutputRequest -> output encoder
| |
warnings.json OutputResult.Warnings
|
appended to final RunOutput only
|
receipt, stderr, final debug warning summary
```
### Attempts And Validation
- `runProducerAttempts` promotes only the terminal accepted or terminal
rejected candidate's module and completed-validator warnings. Operational,
structural, semantic, and module-directed attempts that are superseded are
retained in attempt debug artifacts but not in the final collection.
- A module-directed semantic-reconciliation retry adds its fallback warning
only when no retry remains. Earlier attempt warnings are discarded.
- On `warn_continue`, warnings from the otherwise accepted candidate and
completed approved or rejected validators are retained. One fixed,
non-sensitive `validator_execution_incomplete` warning is added for every
failed validator. Skipped validators affect the validation summary and final
`incomplete` status but do not receive such a warning.
- A semantic terminal rejection retains only warnings from that rejected
attempt. Structural rejection after producer failure cannot retain a
candidate warning because no valid candidate result exists.
These behaviors are protected by the producer-attempt, extract-handoff,
rejection-warning, normalize-retry, and attempt-debug tests. They are the right
foundation for the redesign and should not be replaced with early-exit or
all-attempt accumulation.
### Concurrency And Ordering
Extract jobs are dispatched chunk-first and lane-second. Results are stored by
chunk index, finalized in ascending chunk order, and lane continuations are
merged into a slice indexed by configured lane order. Pipeline steps run in
configured order. The resulting public order is therefore:
1. pre-run reference warnings;
2. chunk-stage warnings;
3. step order;
4. configured lane order within each step;
5. chunk order within extract;
6. merge warnings; then
7. normalize warnings; followed by any output-result warnings.
`TestRunnerBoundsExtractJobsAndStabilizesReverseCompletion` exercises warning
order under reversed completion. No completion-order leak was found.
### Chunk Plans And Checkpoints
- A reusable chunk plan stores producer warnings only. Current validators run
again, and their current warnings are appended. Warnings from a cached plan
candidate that fails current validation are discarded before regeneration.
- Accepted extract, merge, and normalize checkpoints store the terminal
warnings for their stage. Reuse loads and appends those warnings once at the
same logical handoff. Tests compare fresh and resumed warning collections and
preserve their order.
- Validation-incomplete accepted outputs are not reusable, preventing a later
run from silently treating incomplete validation as complete.
- Required accepted-normalize hydration replays that normalize checkpoint's
warnings; checkpoint decisions separately expose that reuse occurred.
The recommended redesign should keep fresh and resumed logical diagnostics
equivalent. Whether a result was reused belongs in checkpoint provenance, not
in the diagnostic grouping key; adding a `reused` distinction would fragment
groups and make equivalent runs present differently.
### Terminal Surfaces
| Surface | Current content | Audience | Audit result |
| --- | --- | --- | --- |
| `RunOutput.Warnings` | Flat final slice | Framework and CLI | Canonical in-memory list, but lacks origin and bounds. |
| Published `warnings.json` | Object containing the warnings passed into the output encoder | Durable consumers | Exact for the production JSON encoder unless the encoder itself returns warnings. |
| `index.json` | Path to `warnings.json` | Durable consumers | Stable discovery path; no separate diagnostic-detail path. |
| Run-result v1 | `warning_count = len(final RunOutput.Warnings)` | Subprocess callers | Count only; no group/occurrence distinction. |
| Human stderr | `run completed with N warning(s)` | Operators | Count only and no direct detail path. Successful exit remains zero. |
| Manifest | Validation and rejection summaries, no warning collection | Durable provenance | Correctly avoids duplicating the flat list. |
| Debug summary `warnings.json` | Raw final warning array | Operators/developers | Includes final output-result warnings and can therefore differ from published `warnings.json`. |
| Debug run report | Final warning count | Operators/developers | Same final slice length as receipt and stderr. |
| Attempt/stage debug | Candidate-local warning detail and origin in path/envelope | Forensics | Sufficient to diagnose provenance, but debug capture is optional and is not a durable consumer contract. |
## Empirical Measurements
### Offline And Synthetic Runs
| Scenario | Result | What it establishes |
| --- | --- | --- |
| Maintained complete D&D config and transcript with the repository's offline fake LLM | Approved, 10 normalized outputs, 0 rejected outputs, 12 warnings; receipt, stderr, and published file all reported 12 | An ordinary structurally successful workflow can be noisy without fallback or incomplete validation. |
| Same complete run, grouped after publication | Three reason codes, seven exact `(reason, scope, message)` tuples, maximum exact-tuple repetition of three | The flat count materially overstates distinct operator conditions. Scope resets within chunks and does not identify origin. |
| Maintained focused scene-description workflow | Approved, one normalized output, 0 warnings | The warning channel can be quiet when synthetic model text is lexically grounded. |
| Generic warning publication contract | One warning reaches successful stderr, durable output, and debug summary | The ordinary pre-output path is consistent. |
| NPC semantic-reconciliation candidate-limit fallback | No LLM call, all records preserved, one exhaustion warning, total warnings no greater than 20 | Fallback is bounded and materially different from routine normalization. |
| `warn_continue` with two failed validators and one skipped validator | Validation status contains all three incomplete validators; warning slice contains two execution-incomplete records | Current warning count does not describe all incomplete validation. |
| Retrying extract candidate | Two producer attempts; only the accepted attempt's one warning is final | Retry does not amplify abandoned warnings. |
| Terminal semantic rejection | Only the final rejected attempt's operation and validator warnings are final | Rejection diagnostics are retained without retaining superseded warnings. |
| Fresh versus reused extract checkpoint | Warning collections are deeply equal | Checkpoint replay does not itself amplify warnings. |
| Spell normalizer with 21 unresolved entries | 20 records: 19 samples plus one omission record saying two additional warnings were omitted | `warning_count` is neither exact occurrence count nor distinct-condition count. |
The focused audit tests passed in `internal/cli`,
`internal/framework/pipeline`, the NPC-registry and spell normalizers, and all
D&D packages.
### Bounded Sample Review
The complete offline D&D run produced:
| Reason | Count | Sample | Review |
| --- | ---: | --- | --- |
| `location_not_near_source` | 4 | `Moon Gate` was absent from cited text | Correctly identifies deliberately unsupported fake output. Three records shared the same exact tuple because chunk and stage origin were lost. |
| `location_occurrence_not_near_source` | 4 | A `Moon Gate` visit was absent from cited text | Correctly identifies the same unsupported registry-driven occurrence, but repeats the same operator concern across extraction and normalization. |
| `scene_description_not_near_source` | 4 | A title or summary had no significant exact token in cited text | Mixed value. Generic `session scene` prose is ungrounded, while `Arrival` versus transcript `arrive` illustrates an expected lexical false positive. |
This fake workflow is an integration fixture, not a model-quality benchmark.
It nonetheless proves that the checks carry useful evidence while being too
imprecise and repetitive to serve as one-warning-per-record operator alerts.
### Production Evidence Still Needed
The reported 85-warning run establishes that high volume occurs in practice,
but the following remain unknown:
- dominant production reason codes and stage/lane sources;
- unique group count versus repeated occurrence count;
- false-positive rate for each relatedness family;
- how much volume comes from normalization observations versus advisories;
- whether fresh and resumed production runs remain equivalent; and
- a defensible numerical acceptance target.
If further data is worthwhile, the operator can supply the v1 receipt,
`warnings.json`, and manifest validation summaries without supplying transcript
or lane artifacts. An initial privacy-preserving report should group by reason
code and normalized scope family, count exact repeated tuples, and omit message
text. Reviewing heuristic precision requires a separately approved bounded
sample with its cited source context.
## Classification Of Current Conditions
| Target disposition | Current families | Result impact | Operator action | Durable placement |
| --- | --- | --- | --- | --- |
| **Warning** | Empty reference; validator failure or skip accepted under `warn_continue`; unavailable required scene classification; exhausted semantic reconciliation | A configured process completed under an allowed degraded or incomplete policy rather than completing normally | Correct reference/configuration, inspect provider/validator, or rerun | Actionable `warnings.json`, receipt summary, stderr summary, debug |
| **Advisory** | Source-relatedness heuristics; unresolved spell or registry membership; guarded invalid semantic proposal; any future LLM-judged uncertainty or extraction-quality signal | Uncertain data quality or poor model proposal, but accepted data is structurally valid and deterministic guards prevented unsafe mutation | Optional model/source review; no routine action for every record | Durable diagnostic detail and debug; never a top-level warning |
| **Observation** | Whitespace/name/ID/source-reference canonicalization; canonical ordering; exact and approved semantic duplicate consolidation | Successful intended normalization | None under normal operation | Durable bounded normalization diagnostics or debug; no stderr warning |
| **Not a diagnostic** | Rejection, invalid structure, cancellation, persistence error, provider failure under fail-run policy | Candidate or run did not complete according to policy | Inspect rejection/error and retry or correct input/configuration | Existing rejection, validation summary, error, and debug contracts |
Exact and semantic duplicate consolidation should remain distinguishable in
category or reason metadata even though both are observations. Semantic
reconciliation exhaustion remains a warning because a capability was not
applied; successful approved consolidation is an observation because it is the
normalizer's intended work.
## Ranked Findings
### AUD-WARN-001 — The Flat Warning Type Destroys Signal Quality
- **Priority:** High operator impact; high implementation leverage.
- **Evidence:** `contracts.Warning` has only scope, reason, and message. Routine
normalizer changes, heuristic doubt, fallback, and incomplete validation all
enter the same slice and the same CLI count. The offline complete run's 12
records were all advisories; the operator observed 85 records in a successful
real run.
- **Impact:** Operators cannot tell whether a warning requires a rerun, a
configuration repair, optional review, or no action. Repeated routine output
trains them to ignore the channel.
- **Recommendation:** Replace the flat semantic contract with explicit
`warning`, `advisory`, and `observation` dispositions plus a small category
vocabulary. Do not infer disposition from message text or require every
downstream consumer to maintain a reason-code policy table.
### AUD-WARN-002 — Warning Records Lose The Origin Needed For Diagnosis And Aggregation
- **Priority:** High correctness and usability impact.
- **Evidence:** The runner knows stage, step, lane, module, validator, chunk ID,
and chunk index at promotion time, but `terminalWarnings` flattens module and
validator records into `[]contracts.Warning`. Per-chunk scopes such as
`locations[0]` and `occurrences[0]` then repeat without identifying their
chunk or producer. `source_references_normalized` is intentionally shared
across families and is therefore especially ambiguous.
- **Impact:** `warnings.json` cannot answer which stage or module produced a
record. Message- or scope-based deduplication would merge unrelated findings
or retain accidental duplicates.
- **Recommendation:** Keep module findings free of framework context, then have
the framework add a structured origin envelope before promotion. Validator
findings must retain validator identity instead of passing through
`validationReport.Warnings()` as a flat slice.
### AUD-WARN-003 — Warning Volume Is Not End-To-End Bounded Or Validated
- **Priority:** High robustness impact; medium immediate likelihood.
- **Evidence:** D&D's `LimitWarnings` caps most individual producers at 20, but
NPC- and spell-relatedness return one warning per record without the helper.
Every extract validator is invoked per chunk, all ten relatedness checks run
again after normalization, and there is no run-level collector. The generic
contract validates neither disposition nor reason/message size, UTF-8,
blankness, or total records.
- **Impact:** Warning memory and output grow with chunks, lanes, configured
validators, and record counts. Local omission records lose exact occurrence
semantics while still incrementing `warning_count`.
- **Recommendation:** Add a generic bounded diagnostic collector that preserves
exact occurrence counts and bounded samples. Validate all diagnostic fields
at the module/framework boundary. Immediately bring NPC and spell
relatedness under the existing cap if the full redesign is staged.
### AUD-WARN-004 — Output Encoder Warnings Make Durable Surfaces Disagree
- **Priority:** Medium current impact; high contract correctness risk.
- **Evidence:** `Runner.Run` passes existing warnings to `encoder.Encode`, then
the production encoder serializes `warnings.json`. Only after encoding does
the runner append `OutputResult.Warnings`. The receipt, stderr, debug summary,
and debug run report see the final slice; the already-created published file
cannot. No production encoder currently returns a warning, so ordinary JSON
runs do not trigger the defect.
- **Impact:** A valid output-module implementation can violate the documented
claim that `warning_count` describes the published warning collection.
- **Recommendation:** Remove successful output warnings from the output-module
contract unless a demonstrated use case requires them; encoding failures
should be errors and optional encoder observations should be debug data. A
two-phase finalize API is the viable but more complex alternative.
### AUD-WARN-005 — Validation Skips Are Incomplete But Not Warned
- **Priority:** Medium operator/correctness impact.
- **Evidence:** `firstIncompleteValidation` treats failed and skipped validators
alike, and validation summaries include both. `incompleteValidationWarnings`
emits records only for `validationFailed`. The focused test demonstrates
three incomplete validators but two warnings.
- **Impact:** A successful run can have `validation_status: incomplete` while
its warning count understates or even omits the affected validators. A caller
that checks only warnings receives a weaker signal than the manifest and
receipt status.
- **Recommendation:** Produce one aggregated incomplete-validation warning
group whose occurrences cover both failure and skip, while retaining typed
outcome and safe reason metadata in the validation summary. Do not expose
provider errors or arbitrary skip prose in model or operator messages.
### AUD-WARN-006 — Relatedness Checks Are Useful But Repetitive And Lexically Weak
- **Priority:** Medium operator impact; low acceptance-policy urgency.
- **Evidence:** Every family runs the advisory in both extract and normalize
chains. The complete fixture contains exact repeated tuples, and the checks
rely on exact normalized token sequences or a minimal significant-token
overlap. Reason naming drifts between `*_not_near_source` and
`*_source_unrelated`.
- **Impact:** The checks can catch unsupported entities, but aliases, pronouns,
inflection, and generic scene prose create predictable false positives. Flat
per-record presentation magnifies them.
- **Recommendation:** Retain the validators and their stage-local execution,
but classify and aggregate them as advisories. Normalize reason-code naming
when the diagnostic contract changes. Do not strengthen them into rejection
rules without a human-reviewed production evaluation.
### AUD-WARN-007 — `warning_count` Has No Stable Operational Meaning
- **Priority:** High downstream-contract impact.
- **Evidence:** The receipt and CLI use `len(output.Warnings)`. One list element
can be an omission summary representing several hidden occurrences; repeated
records can represent the same condition; and skipped validators can be
absent. A 21-occurrence spell test produces a list length of 20.
- **Impact:** The value is neither an exact occurrence count nor a distinct
warning-group count. Consumers cannot set policy or present a trustworthy
summary from it.
- **Recommendation:** Introduce explicit warning-group and warning-occurrence
counts in a versioned receipt. Do not silently redefine the v1 field.
## Recommended Target Contract And Presentation Model
### Diagnostic Model
Use one validated internal diagnostic model with these concepts:
- **disposition:** `warning`, `advisory`, or `observation`;
- **category:** a small enum such as `configuration`, `degradation`,
`validation_incomplete`, `data_quality`, `fallback`, or `normalization`;
- **reason code:** stable semantic identity owned by the producer;
- **origin:** framework-added phase/stage, step ID, lane ID, module key,
validator name, chunk ID, and chunk index when applicable;
- **occurrence count:** exact number of matching findings;
- **samples:** a small deterministic list of bounded scope/message pairs; and
- **omitted sample count:** `occurrence_count - len(samples)`, represented as
metadata rather than another diagnostic record.
Errors and rejected outputs must not become diagnostic dispositions. A warning
means that the run completed under policy despite a process-level degradation
or incomplete configured operation. Advisory and observation dispositions can
describe accepted artifact quality and transformation provenance, but no
LLM-judged extraction-quality signal may be promoted to a warning. Validation
status remains authoritative for approval, rejection, and incomplete
validation.
### Aggregation
The framework runner should own aggregation after it enriches findings with
origin and before public output construction. Modules and validators retain
semantic ownership of disposition, category, reason, scope, and message; they
must not own CLI or file presentation.
The default stable key should be:
```text
disposition + category + reason_code
+ phase/stage + step_id + lane_id + module_key + validator_name
```
Chunk, record scope, and message text belong in samples and must not be part of
the group key. This groups repeated per-chunk findings without merging the same
code across distinct producers or pipeline locations. Group order should be
the first occurrence in the runner's existing canonical order; sample order
should follow the same order. A final canonical sort by the complete origin key
is also viable, but completion timing must never choose either order.
Aggregation must be incremental and bounded. Producers should use a shared
collector that counts every occurrence while retaining only bounded samples;
the framework then merges producer groups without reconstructing counts from
omission prose. A global maximum group count is also required, with overflow
represented by structured aggregate metadata and with actionable groups given
priority over lower dispositions.
### Durable Files
Keep one canonical home for each class:
- `warnings.json` should contain versioned, grouped actionable warnings only;
- a new `diagnostics.json` should contain versioned advisory and observation
groups only, avoiding duplication of warning groups;
- `index.json` should link both files;
- `rejected.json` and manifest validation summaries should retain their current
separate responsibilities; and
- debug bundles should retain candidate-attempt detail plus the final grouped
projections.
This is preferable to keeping all detail in `warnings.json` and filtering only
the CLI: downstream consumers would otherwise continue to receive a semantically
mixed warning contract, and routine observations would still dominate the
durable file.
### CLI And Receipt
For a successful human run with actionable warnings, print a concise summary
such as:
```text
notarius: run completed with 2 warning groups (7 occurrences); details=/.../warnings.json
```
Advisories and observations should not produce the warning line. Their durable
path remains discoverable through `index.json`; a concise non-warning count can
be added to the ordinary success line only if operator testing shows value. An
ordinary successful run with no process degradation should write nothing to
the warning stream even when it publishes quality advisories or normalization
observations.
Create `notarius.run-result.v2` rather than redefining v1. It should expose at
least:
- `warning_group_count`;
- `warning_occurrence_count`; and
- `diagnostic_group_count` for non-warning durable groups.
The receipt should continue to expose validation status, validation summaries,
and rejected-output count independently. Process exit behavior should not
change as part of warning presentation reform.
### Checkpoint Semantics
Store the structured terminal diagnostic groups with accepted checkpoints and
replay them exactly once at their logical stage. Fresh and reused runs should
produce the same public groups and counts. Checkpoint events and debug records,
not diagnostic identity, should disclose whether computation was reused.
### Output Encoder Boundary
Prefer removing `OutputResult.Warnings`. A successful output encoder should
either return the complete logical files or fail. If future encoders genuinely
need to produce durable post-encoding warnings, introduce an explicit
two-phase prepare/finalize contract so those warnings can be included in the
same published collection. Do not retain the current self-inconsistent
one-pass capability.
## Resolution Of Required Design Questions
| Question | Recommendation | Viable alternative and tradeoff |
| --- | --- | --- |
| Explicit severity/disposition or external reason mapping? | Put validated disposition and category in the contract. | A central reason-code registry avoids payload fields but makes new modules depend on a second synchronized policy table and leaves downstream meaning implicit. |
| Keep all detail in `warnings.json` or separate it? | Separate grouped actionable warnings from grouped advisories/observations in `diagnostics.json`. | Keep the flat durable list and aggregate only CLI output; simpler migration, but it preserves the noisy downstream contract and ambiguous count. |
| Who owns aggregation? | Framework runner/coordinator after origin enrichment. | Output module aggregation keeps framework types smaller but duplicates policy across encoders and cannot repair missing validator origin. |
| Stable aggregation key? | Disposition, category, reason code, and full producer origin; exclude chunk/scope/message. | Explicit producer-supplied grouping keys offer flexibility but add another identity that can drift from reason codes. Message-template grouping is brittle and unsafe. |
| Samples and omissions? | Exact occurrence count plus deterministic bounded samples and numeric omitted-sample count. | Omission warning records preserve the current representation but inflate group counts and require prose parsing. |
| `warning_count` semantics? | Version receipt and replace ambiguity with group and occurrence counts. | Keep v1 count as published record length and add optional fields; compatible, but two competing warning counts remain easy to misuse. |
| Which normalization changes remain warnings? | Only exhausted process fallback. Unresolved membership is a data-quality advisory; successful canonicalization, reordering, ID repair, source-ref dedupe, and duplicate consolidation are observations. | Treat unresolved membership or semantic duplicate consolidation as warnings because they affect grounding or cardinality; more conservative, but it violates the process-only warning rule and reports accepted artifact quality as an operational failure. |
| Source-relatedness disposition? | Grouped advisory by default; preserve current approve behavior. | Retain warning disposition or make rejection configurable. Rejection requires production precision evidence; current lexical rules are not strong enough. |
| Checkpoint-loaded warnings? | Present the same logical groups as fresh execution and use checkpoint events for reuse provenance. | Mark groups as replayed; aids forensics but fragments aggregation and makes semantically equivalent runs differ. |
| ADR and schema versions? | Add an ADR and version the run receipt and diagnostic files. | Treat the work as CLI-only presentation and avoid an ADR; insufficient because module contracts, output files, checkpoint payloads, and downstream fields change. |
## Compatibility, Documentation, And ADR Implications
The target alters public and internal contracts enough to require a new ADR.
It should record:
- the distinction among warnings, advisories, observations, rejections, and
errors;
- the invariant that warnings are process-level signals, LLM-judged extraction
quality is never a warning, and ordinary non-degraded success has zero
warnings;
- module semantic ownership versus framework origin/aggregation ownership;
- bounded group and sample semantics;
- fresh/checkpoint equivalence; and
- the output-encoder decision.
Implementation should introduce `notarius.run-result.v2`. The grouped warning
and diagnostic envelopes should each carry their own schema version. Because
the content of `warnings.json` changes incompatibly from a flat array wrapper
to groups, release notes and the published JSON integration contract must call
out the migration. `index.json` gains the diagnostic file path.
Canonical documentation updates belong in:
- `docs/cli.md` for stderr presentation only;
- `docs/operations.md` for operator review and debug workflow;
- `docs/integrations/json-output.md` for warning and diagnostic file schemas;
- `docs/integrations/run-result.md` for v2 fields and compatibility;
- `docs/consumers/subprocess.md` and `docs/consumers/dnd-pipeline.md` for
downstream policy checks;
- `docs/internal/pipeline.md` for promotion, aggregation, retry, and checkpoint
mechanics;
- `docs/internal/modules.md` and `docs/internal/dnd.md` for producer rules and
the D&D classification matrix; and
- `docs/policy/architecture.md` for the durable ownership invariant after the
ADR is accepted and implemented.
No configuration knob is required for the first implementation. A fixed,
well-documented taxonomy is easier to reason about than per-reason display
overrides. Configurable escalation or suppression can be considered only after
production review demonstrates a concrete operator need.
## Test Coverage Assessment
Existing coverage worth preserving includes:
- accepted-attempt and terminal-rejection warning promotion;
- validator failure retry exhaustion and `warn_continue`;
- module semantic retry fallback;
- deterministic warning order under concurrent lane completion;
- chunk-plan invalidation and discarded-cache warning behavior;
- fresh/checkpoint warning equivalence;
- local D&D warning caps and safe dynamic-message quoting;
- JSON warning-file publication; and
- CLI stderr, debug, and receipt counts.
Material gaps are:
- no bound test for NPC- or spell-relatedness warnings;
- no generic warning-field or result-size validation;
- no test for output encoder warnings versus published `warnings.json`;
- no operator-level aggregation or bounded-sample tests;
- no fresh/resume test for grouped counts because groups do not yet exist; and
- no production evaluation of advisory precision.
Tests should protect the semantic relationships: exact occurrence counts,
bounded samples, deterministic group order, actionable-only warning
presentation, and cross-surface equality. They should not assert one exact
warning count for every complete D&D run or treat message wording as a public
API unless the wording itself enforces a security boundary.
## Audit Conclusion
The application is in a good position for warning reform. Its retry,
validation, checkpoint, and concurrency mechanics provide reliable points at
which to attach structured diagnostics. The most valuable change is not to
suppress individual reason codes; it is to replace the semantically flat,
origin-free collection with bounded typed groups and to reserve the word
“warning” for conditions that merit operator attention.
Provider-backed runs would improve prioritization and help tune the D&D
advisories, but they are not necessary to conclude that routine normalization
and heuristic doubt should not dominate stderr or the durable warning
contract. They should be gathered before changing heuristic acceptance policy
or adopting a numerical production warning-volume target.

View File

@@ -10,77 +10,13 @@ not as committed release dates.
PromptKit now owns structural output repair within one completion. Notarius PromptKit now owns structural output repair within one completion. Notarius
owns stage candidates, validator chains, semantic rejection policy, bounded owns stage candidates, validator chains, semantic rejection policy, bounded
feedback-aware stage retries, validation provenance, and reusable-state feedback-aware stage retries, validation provenance, and reusable-state
eligibility. The remaining near-term work applies those completed foundations eligibility, and the separation of actionable process warnings from quality
to domain review and operator-facing diagnostics. diagnostics. The remaining near-term work applies those completed foundations
to domain review and empirical evaluation.
### D&D Combat Scene Semantic Validation Near-term reliability work should now be selected from the concrete evaluation
and extension opportunities below. The retry, validation, and subprocess
- Add an optional production LLM-backed D&D validator that determines whether foundations described above are implemented current behavior.
proposed scene boundaries and classifications represent substantive active
combat correctly. Its central quality goal is that active combat is kept in
coherent scenes classified as `combat`, rather than split incorrectly or
hidden inside scenes classified as `narrative`, `recap`, or `meta`.
- Resolve the validator's exact target before implementation. The current
`dnd/scenes` chunker owns only complete, gap-free source ranges, while the
per-chunk `dnd/scene-descriptions` extractor owns the `combat`, `narrative`,
`recap`, and `meta` classification. The preferred initial placement is
therefore an extract-stage validator for `dnd/scene-descriptions`, where it
can compare one proposed kind with the corresponding transcript chunk.
- Consider a chunk-stage LLM validator only for a distinct boundary-coherence
question that can be answered from the complete transcript and proposed
range map, such as whether one continuous combat was fragmented across
inappropriate scene boundaries. Do not duplicate the same classification
judgment at both stages. Moving classification into chunk-plan annotations
would change the deliberately minimal, annotation-free chunk contract and
requires an explicit architecture review before it is selected.
- Validate both false negatives and false positives: a non-combat kind must not
omit substantive active combat, and a combat kind must be supported by such
combat. Keep the existing deterministic downstream rule that combat-turn
extraction runs only for an exact `combat` scene classification; semantic
review improves the upstream classification but does not replace that gate.
- Run the semantic validator through PromptKit, use a minimal required-field
structured response schema, and let PromptKit repair structural validator
output within its bounded budget. A contract-invalid final validator response
is a validator execution failure, not a semantic rejection and not a reason
to recursively validate the validator.
- Evaluate the prompt and decision policy against a small human-reviewed set
containing combat setup, active turns, interruptions, multi-phase encounters,
brief rules discussion, aftermath, recalled combat, and false-positive
hostile dialogue. Measure false acceptance, false rejection, retry success,
added calls, latency, and token cost before placing it in the production
default chain.
- An ADR is not required if classification remains owned by
`dnd/scene-descriptions` and the validator follows the generic validation ADR.
Create or supersede an ADR if the work transfers scene classification into
the chunker or otherwise changes stage ownership or the durable chunk-plan
contract.
### Warning Signal And Presentation Reform
- Audit every warning producer and representative successful runs. Ordinary
success producing dozens of warnings is a failed operator experience: the
volume obscures actionable problems and trains operators to ignore the
warning channel.
- Define a small warning taxonomy that distinguishes actionable degradation,
incomplete validation, lossy fallback, and data-quality risk from routine
normalization observations or informational diagnostics. Preserve detailed
traceability in debug or manifest data without promoting every observation
to a top-level CLI warning.
- Consider stable deduplication and aggregation by scope and reason code,
bounded samples plus omitted counts, and a concise CLI summary with a path to
detailed diagnostics. Do not suppress genuine validator execution failures
merely to reduce the count.
- Decide which warnings affect process status, rejection summaries, durable run
receipts, or only debug output. Ensure warning ordering and aggregation are
deterministic across concurrent execution.
- Establish a representative warning-volume acceptance target and human review
workflow before changing individual producers piecemeal. The intended result
is not zero warnings; it is a small set in which every surfaced warning merits
operator attention.
- This work does not require an ADR unless it changes validation acceptance,
failure, or durable contract semantics. CLI presentation and diagnostic
taxonomy otherwise belong in a feature roadmap followed by updates to their
canonical configuration, operations, integration, and internal documents.
## Near-Term D&D Pipeline ## Near-Term D&D Pipeline

View File

@@ -71,28 +71,39 @@ func TestAssembledSpellPipelineNormalizesMergedCasts(t *testing.T) {
t.Fatalf("distinct cast = %#v, want separate evidence event", distinct) t.Fatalf("distinct cast = %#v, want separate evidence event", distinct)
} }
wantWarningReasons := []string{ wantDiagnosticReasons := []string{
spellnormalize.ReasonCodeSpellNameCanonicalized, spellnormalize.ReasonCodeSpellNameCanonicalized,
spellnormalize.ReasonCodeSourceReferencesNormalized, spellnormalize.ReasonCodeSourceReferencesNormalized,
spellnormalize.ReasonCodeDuplicateSpellCastCollapsed, spellnormalize.ReasonCodeDuplicateSpellCastCollapsed,
"spell_not_near_source", "spell_not_near_source",
} }
gotWarningReasons := make([]string, len(output.Warnings)) gotDiagnosticReasons := make([]string, len(output.Diagnostics.Groups))
for index, warning := range output.Warnings { for index, group := range output.Diagnostics.Groups {
gotWarningReasons[index] = warning.ReasonCode gotDiagnosticReasons[index] = group.ReasonCode
} }
if !reflect.DeepEqual(gotWarningReasons, wantWarningReasons) { if !reflect.DeepEqual(gotDiagnosticReasons, wantDiagnosticReasons) {
t.Fatalf("warnings = %#v, want deterministic normalize and validation warnings", output.Warnings) t.Fatalf("diagnostics = %#v, want deterministic normalize and validation diagnostics", output.Diagnostics)
} }
if output.Warnings[2].Scope != "spell_casts[0]" || !strings.Contains(output.Warnings[2].Message, "retained input index 0") || !strings.Contains(output.Warnings[2].Message, "removed input indices [1]") { if output.Diagnostics.Groups[2].Samples[0].Scope != "spell_casts[0]" || !strings.Contains(output.Diagnostics.Groups[2].Samples[0].Message, "retained input index 0") || !strings.Contains(output.Diagnostics.Groups[2].Samples[0].Message, "removed input indices [1]") {
t.Fatalf("duplicate warning = %#v, want retained and removed merged indices", output.Warnings[2]) t.Fatalf("duplicate diagnostic = %#v, want retained and removed merged indices", output.Diagnostics.Groups[2])
} }
warningsFile := decodeAssembledOutput[struct { warningsFile := decodeAssembledOutput[struct {
Warnings []contracts.Warning `json:"warnings"` Groups []contracts.DiagnosticGroup `json:"groups"`
}](t, output.OutputFiles, "warnings.json") }](t, output.OutputFiles, "warnings.json")
if !reflect.DeepEqual(warningsFile.Warnings, output.Warnings) { if len(warningsFile.Groups) != 0 {
t.Fatalf("warnings file = %#v, run warnings = %#v, want manifest output path to preserve warnings", warningsFile.Warnings, output.Warnings) t.Fatalf("warnings file = %#v, want no process warnings for advisory-only diagnostics", warningsFile.Groups)
}
diagnosticsFile := decodeAssembledOutput[struct {
SchemaVersion string `json:"schema_version"`
GroupCount int `json:"group_count"`
OccurrenceCount int `json:"occurrence_count"`
Truncated bool `json:"truncated"`
UnrepresentedOccurrenceCount int `json:"unrepresented_occurrence_count"`
Groups []contracts.DiagnosticGroup `json:"groups"`
}](t, output.OutputFiles, "diagnostics.json")
if diagnosticsFile.SchemaVersion != "notarius.diagnostics.v1" || diagnosticsFile.GroupCount != len(output.Diagnostics.Groups) || !reflect.DeepEqual(diagnosticsFile.Groups, output.Diagnostics.Groups) || diagnosticsFile.OccurrenceCount != diagnosticGroupOccurrences(output.Diagnostics.Groups)+output.Diagnostics.UnrepresentedOccurrenceCount || diagnosticsFile.Truncated != output.Diagnostics.Truncated || diagnosticsFile.UnrepresentedOccurrenceCount != output.Diagnostics.UnrepresentedOccurrenceCount {
t.Fatalf("diagnostics file = %#v, run diagnostics = %#v", diagnosticsFile, output.Diagnostics)
} }
manifest := decodeAssembledOutput[artifacts.RunManifest](t, output.OutputFiles, "manifest.json") manifest := decodeAssembledOutput[artifacts.RunManifest](t, output.OutputFiles, "manifest.json")
if len(manifest.ArtifactLanes) != 1 || manifest.ArtifactLanes[0].Normalizer != spellnormalize.Key { if len(manifest.ArtifactLanes) != 1 || manifest.ArtifactLanes[0].Normalizer != spellnormalize.Key {
@@ -155,14 +166,14 @@ func TestAssembledSpellPipelineHonorsNormalizeValidatorOverride(t *testing.T) {
if err != nil || output.Manifest.ValidationStatus != "approved" || len(output.Rejected) != 0 || len(output.NormalizeOutputs) != 1 { if err != nil || output.Manifest.ValidationStatus != "approved" || len(output.Rejected) != 0 || len(output.NormalizeOutputs) != 1 {
t.Fatalf("Run() error = %v output = %#v, want approved override run", err, output) t.Fatalf("Run() error = %v output = %#v, want approved override run", err, output)
} }
for _, warning := range output.Warnings { for _, group := range output.Diagnostics.Groups {
if warning.ReasonCode == "spell_not_near_source" { if group.ReasonCode == "spell_not_near_source" {
t.Fatalf("warnings = %#v, want explicit validator override to replace default relatedness chain", output.Warnings) t.Fatalf("diagnostics = %#v, want explicit validator override to replace default relatedness chain", output.Diagnostics)
} }
} }
} }
func TestAssembledSpellPipelinePromotesTerminalUnknownSpellWarning(t *testing.T) { func TestAssembledSpellPipelinePromotesTerminalUnknownSpellDiagnostics(t *testing.T) {
registries, resolved, _ := assembledSpellPipeline(t, assembledSpellPipelineOptions{unknownSpell: true}) registries, resolved, _ := assembledSpellPipeline(t, assembledSpellPipelineOptions{unknownSpell: true})
resolved.Steps[0].ArtifactLanes[0].NormalizeValidationPolicy.SemanticRejection = pipeline.SemanticRejectionRejectOutput resolved.Steps[0].ArtifactLanes[0].NormalizeValidationPolicy.SemanticRejection = pipeline.SemanticRejectionRejectOutput
prepared, err := pipeline.Prepare(resolved, registries, pipeline.ModuleDependencies{}) prepared, err := pipeline.Prepare(resolved, registries, pipeline.ModuleDependencies{})
@@ -190,12 +201,12 @@ func TestAssembledSpellPipelinePromotesTerminalUnknownSpellWarning(t *testing.T)
if !reflect.DeepEqual(rejectedFile.Rejected, output.Rejected) { if !reflect.DeepEqual(rejectedFile.Rejected, output.Rejected) {
t.Fatalf("rejected file = %#v, run rejections = %#v, want durable rejection diagnostic", rejectedFile.Rejected, output.Rejected) t.Fatalf("rejected file = %#v, run rejections = %#v, want durable rejection diagnostic", rejectedFile.Rejected, output.Rejected)
} }
if len(output.Warnings) != 2 || output.Warnings[0].ReasonCode != spellnormalize.ReasonCodeSpellNameUnresolved || output.Warnings[0].Scope != "spell_casts[0]" || output.Warnings[1].ReasonCode != "spell_not_near_source" { if len(output.Diagnostics.Groups) != 2 || output.Diagnostics.Groups[0].ReasonCode != spellnormalize.ReasonCodeSpellNameUnresolved || output.Diagnostics.Groups[0].Samples[0].Scope != "spell_casts[0]" || output.Diagnostics.Groups[1].ReasonCode != "spell_not_near_source" {
t.Fatalf("warnings = %#v, want complete terminal normalize validation warnings", output.Warnings) t.Fatalf("diagnostics = %#v, want complete terminal normalize validation diagnostics", output.Diagnostics)
} }
} }
func TestAssembledSpellPipelinePromotesUnknownSpellWarningWhenOverrideAccepts(t *testing.T) { func TestAssembledSpellPipelinePromotesUnknownSpellAdvisoryWhenOverrideAccepts(t *testing.T) {
registries, resolved, _ := assembledSpellPipeline(t, assembledSpellPipelineOptions{normalizeValidatorOverride: true, unknownSpell: true}) registries, resolved, _ := assembledSpellPipeline(t, assembledSpellPipelineOptions{normalizeValidatorOverride: true, unknownSpell: true})
prepared, err := pipeline.Prepare(resolved, registries, pipeline.ModuleDependencies{}) prepared, err := pipeline.Prepare(resolved, registries, pipeline.ModuleDependencies{})
if err != nil { if err != nil {
@@ -219,14 +230,20 @@ func TestAssembledSpellPipelinePromotesUnknownSpellWarningWhenOverrideAccepts(t
if len(normalized.SpellCasts) != 1 || normalized.SpellCasts[0].Spell != "Mysterious Burst" { if len(normalized.SpellCasts) != 1 || normalized.SpellCasts[0].Spell != "Mysterious Burst" {
t.Fatalf("normalized casts = %#v, want unresolved name preserved", normalized.SpellCasts) t.Fatalf("normalized casts = %#v, want unresolved name preserved", normalized.SpellCasts)
} }
if len(output.Warnings) != 1 || output.Warnings[0].ReasonCode != spellnormalize.ReasonCodeSpellNameUnresolved || output.Warnings[0].Scope != "spell_casts[0]" { if len(output.Diagnostics.Groups) != 1 || output.Diagnostics.Groups[0].ReasonCode != spellnormalize.ReasonCodeSpellNameUnresolved || output.Diagnostics.Groups[0].Samples[0].Scope != "spell_casts[0]" {
t.Fatalf("warnings = %#v, want promoted scoped unresolved-name warning", output.Warnings) t.Fatalf("diagnostics = %#v, want promoted scoped unresolved-name diagnostic", output.Diagnostics)
} }
warningsFile := decodeAssembledOutput[struct { warningsFile := decodeAssembledOutput[struct {
Warnings []contracts.Warning `json:"warnings"` Groups []contracts.DiagnosticGroup `json:"groups"`
}](t, output.OutputFiles, "warnings.json") }](t, output.OutputFiles, "warnings.json")
if !reflect.DeepEqual(warningsFile.Warnings, output.Warnings) { if len(warningsFile.Groups) != 0 {
t.Fatalf("warnings file = %#v, run warnings = %#v, want durable unresolved-name warning", warningsFile.Warnings, output.Warnings) t.Fatalf("warnings file = %#v, want no process warnings for an advisory diagnostic", warningsFile.Groups)
}
diagnosticsFile := decodeAssembledOutput[struct {
Groups []contracts.DiagnosticGroup `json:"groups"`
}](t, output.OutputFiles, "diagnostics.json")
if !reflect.DeepEqual(diagnosticsFile.Groups, output.Diagnostics.Groups) {
t.Fatalf("diagnostics file = %#v, run diagnostics = %#v", diagnosticsFile.Groups, output.Diagnostics)
} }
} }
@@ -429,6 +446,14 @@ func (e *assembledSpellExtractor) chunkIndexesSnapshot() []int {
return append([]int(nil), e.chunkIndexes...) return append([]int(nil), e.chunkIndexes...)
} }
func diagnosticGroupOccurrences(groups []contracts.DiagnosticGroup) int {
count := 0
for _, group := range groups {
count += group.OccurrenceCount
}
return count
}
func decodeAssembledOutput[T any](t *testing.T, files []contracts.OutputFile, name string) T { func decodeAssembledOutput[T any](t *testing.T, files []contracts.OutputFile, name string) T {
t.Helper() t.Helper()
for _, file := range files { for _, file := range files {

View File

@@ -91,8 +91,8 @@ func TestProductionSceneDescriptionWorkflow(t *testing.T) {
if !reflect.DeepEqual(durable, want) { if !reflect.DeepEqual(durable, want) {
t.Fatalf("durable output payload = %#v, want %#v", durable, want) t.Fatalf("durable output payload = %#v, want %#v", durable, want)
} }
if len(output.Warnings) != 0 { if len(output.Diagnostics.Groups) != 0 {
t.Fatalf("warnings = %#v, want grounded descriptions without warnings", output.Warnings) t.Fatalf("diagnostics = %#v, want grounded descriptions without diagnostics", output.Diagnostics)
} }
} }

View File

@@ -239,10 +239,10 @@ func TestMaintainedMinimalInvocationProducesJSONBundle(t *testing.T) {
t.Fatalf("rejected = %#v, want empty rejection list", rejected.Rejected) t.Fatalf("rejected = %#v, want empty rejection list", rejected.Rejected)
} }
warnings := readProductionJSON[struct { warnings := readProductionJSON[struct {
Warnings []json.RawMessage `json:"warnings"` Groups []json.RawMessage `json:"groups"`
}](t, filepath.Join(runRoot, "warnings.json")) }](t, filepath.Join(runRoot, "warnings.json"))
if len(warnings.Warnings) != 0 { if len(warnings.Groups) != 0 {
t.Fatalf("warnings = %#v, want empty warning list", warnings.Warnings) t.Fatalf("warnings = %#v, want empty warning list", warnings.Groups)
} }
} }

View File

@@ -852,10 +852,10 @@ func TestProductionSceneRunRecordsAnnotationFreeChunkPlanAndProvenance(t *testin
t.Fatalf("chunk map range annotations = %#v, want none", chunkMap.Chunks[0].Annotations) t.Fatalf("chunk map range annotations = %#v, want none", chunkMap.Chunks[0].Annotations)
} }
warnings := readProductionJSON[struct { warnings := readProductionJSON[struct {
Warnings []contracts.Warning `json:"warnings"` Groups []contracts.DiagnosticGroup `json:"groups"`
}](t, filepath.Join(outputRoot, productionRunID, "warnings.json")) }](t, filepath.Join(outputRoot, productionRunID, "warnings.json"))
if len(warnings.Warnings) != 0 { if len(warnings.Groups) != 0 {
t.Fatalf("warnings = %#v, want none", warnings.Warnings) t.Fatalf("warnings = %#v, want none", warnings.Groups)
} }
if len(fake.requestsFor(scenes.PromptID)) != 1 || len(fake.requestsFor(spells.PromptID)) != 1 || len(fake.requestsFor(itemoccurrenceextract.PromptID)) != 1 { if len(fake.requestsFor(scenes.PromptID)) != 1 || len(fake.requestsFor(spells.PromptID)) != 1 || len(fake.requestsFor(itemoccurrenceextract.PromptID)) != 1 {
t.Fatalf("fake prompt requests = %#v, want one scene, spell, and item-occurrence request", fake.requestPrompts()) t.Fatalf("fake prompt requests = %#v, want one scene, spell, and item-occurrence request", fake.requestPrompts())

View File

@@ -4,6 +4,7 @@ import (
"bytes" "bytes"
"os" "os"
"path/filepath" "path/filepath"
"slices"
"strings" "strings"
"testing" "testing"
@@ -16,18 +17,14 @@ func TestReferenceSelectorsParseAndApplyAllDocumentedForms(t *testing.T) {
tests := []struct { tests := []struct {
name string name string
selector string selector string
only []string want []string
wantStage pipeline.ModuleStage
wantLane string
wantSlot string
}{ }{
{name: "flat", selector: "alpha-slot", wantStage: pipeline.StageExtract, wantLane: "alpha", wantSlot: "alpha-slot"}, {name: "pipeline", selector: "shared", want: []string{"alpha.extract.shared", "alpha.merge.shared", "alpha.normalize.shared", "beta.extract.shared", "beta.merge.shared", "beta.normalize.shared"}},
{name: "chunk", selector: "chunk.chunk-slot", wantStage: pipeline.StageChunk, wantSlot: "chunk-slot"}, {name: "chunk", selector: "chunk.chunk-slot", want: []string{"chunk.chunk-slot"}},
{name: "merge", selector: "merge.alpha-merge", only: []string{"alpha"}, wantStage: pipeline.StageMerge, wantLane: "alpha", wantSlot: "alpha-merge"}, {name: "lane", selector: "alpha.shared", want: []string{"alpha.extract.shared", "alpha.merge.shared", "alpha.normalize.shared"}},
{name: "lane", selector: "alpha.alpha-slot", wantStage: pipeline.StageExtract, wantLane: "alpha", wantSlot: "alpha-slot"}, {name: "lane extract", selector: "alpha.extract.alpha-slot", want: []string{"alpha.extract.alpha-slot"}},
{name: "lane extract", selector: "alpha.extract.alpha-slot", wantStage: pipeline.StageExtract, wantLane: "alpha", wantSlot: "alpha-slot"}, {name: "lane merge", selector: "alpha.merge.alpha-merge", want: []string{"alpha.merge.alpha-merge"}},
{name: "lane merge", selector: "alpha.merge.alpha-merge", wantStage: pipeline.StageMerge, wantLane: "alpha", wantSlot: "alpha-merge"}, {name: "lane normalize", selector: "alpha.normalize.alpha-normalize", want: []string{"alpha.normalize.alpha-normalize"}},
{name: "lane normalize", selector: "alpha.normalize.alpha-normalize", wantStage: pipeline.StageNormalize, wantLane: "alpha", wantSlot: "alpha-normalize"},
} }
for _, tt := range tests { for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) { t.Run(tt.name, func(t *testing.T) {
@@ -37,70 +34,135 @@ func TestReferenceSelectorsParseAndApplyAllDocumentedForms(t *testing.T) {
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
overrides, _, err := resolveCLIReferenceRequests(cfg, "demo", tt.only, catalog, []cliReferenceRequest{{Selector: selector, Source: "reference.txt"}}, nil) overrides, _, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog, []cliReferenceRequest{{Selector: selector, Source: "reference.txt"}}, nil)
if err != nil { if err != nil {
t.Fatalf("resolve selector: %v", err) t.Fatalf("resolve selector: %v", err)
} }
if len(overrides) != 1 { if got := referenceContractBindingLabels(overrides); !slices.Equal(got, tt.want) {
t.Fatalf("overrides = %#v, want one binding", overrides) t.Fatalf("binding targets = %#v, want %#v", got, tt.want)
}
for _, binding := range overrides {
if binding.Source != "reference.txt" || binding.BindingSource != contracts.ReferenceBindingSourceCLI {
t.Fatalf("binding = %#v, want CLI source", binding)
} }
got := overrides[0]
if got.Stage != tt.wantStage || got.LaneID != tt.wantLane || got.SlotName != tt.wantSlot || got.BindingSource != contracts.ReferenceBindingSourceCLI {
t.Fatalf("binding = %#v, want %s/%s/%s from CLI", got, tt.wantStage, tt.wantLane, tt.wantSlot)
} }
}) })
} }
} }
func TestReferenceSelectorsRejectAmbiguityWithSpecificSuggestions(t *testing.T) { func TestReferenceSelectorSpecificityAndFinalOccurrenceChooseConcreteBindings(t *testing.T) {
cfg := referenceContractConfig() cfg := referenceContractConfig()
catalog := referenceContractCatalog(t, true, true) catalog := referenceContractCatalog(t, true, true)
for _, tt := range []struct { requests := []cliReferenceRequest{
name string {Selector: mustParseReferenceSelector(t, "shared", "--reference"), Source: "pipeline-first.txt"},
selector string {Selector: mustParseReferenceSelector(t, "shared", "--reference"), Source: "pipeline-final.txt"},
want []string {Selector: mustParseReferenceSelector(t, "alpha.shared", "--reference"), Source: "lane.txt"},
}{ {Selector: mustParseReferenceSelector(t, "alpha.extract.shared", "--reference"), Source: "binding.txt"},
{name: "flat shared slot", selector: "shared", want: []string{"alpha.extract.shared", "beta.extract.shared"}}, }
{name: "lane shared slot", selector: "alpha.shared", want: []string{"alpha.extract.shared", "alpha.merge.shared", "alpha.normalize.shared"}}, overrides, unbinds, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog, requests, nil)
{name: "all mergers", selector: "merge.shared", want: []string{"alpha.merge.shared", "beta.merge.shared"}},
} {
t.Run(tt.name, func(t *testing.T) {
selector, err := parseReferenceSelector(tt.selector, "--reference")
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
_, _, err = resolveCLIReferenceRequests(cfg, "demo", nil, catalog, []cliReferenceRequest{{Selector: selector, Source: "reference.txt"}}, nil) if len(unbinds) != 0 {
if err == nil { t.Fatalf("unbinds = %#v, want none", unbinds)
t.Fatal("resolve selector succeeded, want ambiguity error")
} }
for _, fragment := range tt.want { want := map[string]string{
if !strings.Contains(err.Error(), fragment) { "alpha.extract.shared": "binding.txt",
t.Fatalf("error = %q, want suggestion %q", err, fragment) "alpha.merge.shared": "lane.txt",
"alpha.normalize.shared": "lane.txt",
"beta.extract.shared": "pipeline-final.txt",
"beta.merge.shared": "pipeline-final.txt",
"beta.normalize.shared": "pipeline-final.txt",
} }
for _, binding := range overrides {
label := referenceContractBindingLabel(binding)
if binding.Source != want[label] {
t.Fatalf("binding %s source = %q, want %q", label, binding.Source, want[label])
} }
}) delete(want, label)
}
if len(want) != 0 {
t.Fatalf("missing bindings: %#v", want)
} }
} }
func TestReferenceSelectorsRespectSelectedLanesBeforeMaterialization(t *testing.T) { func TestCompleteDNDSharedCLIReferencesExpandAcrossCompatibleTargets(t *testing.T) {
cfg := loadMaintainedExample(t, repositoryPath("examples", "dnd-complete.config.yml"))
catalog := catalogFromRegistries(productionTestComponents(t).registries)
sources := map[string]string{
"party": "/references/party.txt",
"players": "/references/players.txt",
"glossary": "/references/glossary.txt",
"spell_catalog": "/references/spells.json",
}
requests := make([]cliReferenceRequest, 0, len(sources))
for _, slot := range []string{"party", "players", "glossary", "spell_catalog"} {
requests = append(requests, cliReferenceRequest{
Selector: mustParseReferenceSelector(t, slot, "--reference"),
Source: sources[slot],
})
}
overrides, unbinds, err := resolveCLIReferenceRequests(cfg, "dnd-session", nil, catalog, requests, nil)
if err != nil {
t.Fatalf("expand complete D&D references: %v", err)
}
if len(unbinds) != 0 {
t.Fatalf("unbinds = %#v, want none", unbinds)
}
actual := make(map[string]pipeline.ReferenceBinding, len(overrides))
for _, binding := range overrides {
actual[referenceContractBindingLabel(binding)] = binding
}
targets, err := selectedReferenceTargets(cfg, "dnd-session", nil, catalog)
if err != nil {
t.Fatal(err)
}
matched := make(map[string]int, len(sources))
for _, target := range targets {
for slot, sourcePath := range sources {
if _, ok := target.slots[slot]; !ok {
continue
}
matched[slot]++
label := targetLabel(target) + "." + slot
binding, ok := actual[label]
if !ok || binding.Source != sourcePath || binding.BindingSource != contracts.ReferenceBindingSourceCLI {
t.Fatalf("binding %q = %#v, want CLI source %q", label, binding, sourcePath)
}
}
}
for slot := range sources {
if matched[slot] < 2 {
t.Fatalf("reference %q matched %d target(s), want a shared D&D reference", slot, matched[slot])
}
}
if _, err := cfg.Resolve(config.ResolveInput{PipelineID: "dnd-session", Catalog: catalog, ReferenceOverrides: overrides}); err != nil {
t.Fatalf("resolve complete D&D CLI references: %v", err)
}
}
func TestReferenceSelectorsRejectInvalidOrUnselectedScopesBeforeMaterialization(t *testing.T) {
cfg := referenceContractConfig() cfg := referenceContractConfig()
catalog := referenceContractCatalog(t, true, true) catalog := referenceContractCatalog(t, true, true)
for _, tt := range []struct { for _, tt := range []struct {
name string name string
selector string selector string
only []string
want string want string
}{ }{
{name: "unselected lane", selector: "beta.extract.beta-slot", want: `reference lane "beta" is not selected`}, {name: "pipeline slot", selector: "missing", want: `reference slot "missing" is not declared by any selected target`},
{name: "lane slot", selector: "alpha.missing", want: `reference slot "missing" is not declared by selected lane "alpha"`},
{name: "binding slot", selector: "alpha.extract.missing", want: `reference slot "missing" is not declared`},
{name: "former merge shorthand", selector: "merge.shared", want: `reference lane "merge" is not selected`},
{name: "unselected lane", selector: "beta.extract.beta-slot", only: []string{"alpha"}, want: `reference lane "beta" is not selected`},
{name: "unknown lane", selector: "missing.extract.beta-slot", want: `reference lane "missing" is not selected`}, {name: "unknown lane", selector: "missing.extract.beta-slot", want: `reference lane "missing" is not selected`},
} { } {
t.Run(tt.name, func(t *testing.T) { t.Run(tt.name, func(t *testing.T) {
selector, err := parseReferenceSelector(tt.selector, "--reference") selector := mustParseReferenceSelector(t, tt.selector, "--reference")
if err != nil { _, _, err := resolveCLIReferenceRequests(cfg, "demo", tt.only, catalog, []cliReferenceRequest{{Selector: selector, Source: filepath.Join(t.TempDir(), "missing.txt")}}, nil)
t.Fatal(err)
}
_, _, err = resolveCLIReferenceRequests(cfg, "demo", []string{"alpha"}, catalog, []cliReferenceRequest{{Selector: selector, Source: filepath.Join(t.TempDir(), "missing.txt")}}, nil)
if err == nil || !strings.Contains(err.Error(), tt.want) || strings.Contains(err.Error(), "missing.txt") { if err == nil || !strings.Contains(err.Error(), tt.want) || strings.Contains(err.Error(), "missing.txt") {
t.Fatalf("error = %v, want selection failure before file access", err) t.Fatalf("error = %v, want selection failure containing %q before file access", err, tt.want)
} }
}) })
} }
@@ -131,40 +193,50 @@ func TestReferenceSyntaxErrorsReturnTwo(t *testing.T) {
} }
} }
func TestReferenceOverridesUseFinalExactTargetBinding(t *testing.T) { func TestReferenceBindAndUnbindSpecificity(t *testing.T) {
cfg := referenceContractConfig() cfg := referenceContractConfig()
catalog := referenceContractCatalog(t, true, true) catalog := referenceContractCatalog(t, true, true)
alphaShared, err := parseReferenceSelector("alpha.extract.shared", "--reference") t.Run("specific unbind carves out broad binding", func(t *testing.T) {
overrides, unbinds, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog,
[]cliReferenceRequest{{Selector: mustParseReferenceSelector(t, "shared", "--reference"), Source: "shared.txt"}},
[]cliReferenceUnbindRequest{{Selector: mustParseReferenceSelector(t, "alpha.extract.shared", "--without-reference")}},
)
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
betaShared, err := parseReferenceSelector("beta.extract.shared", "--reference") if got := referenceContractBindingLabels(overrides); slices.Contains(got, "alpha.extract.shared") || len(got) != 5 {
t.Fatalf("overrides = %#v, want all shared targets except alpha extract", got)
}
if got := referenceContractUnbindLabels(unbinds); !slices.Equal(got, []string{"alpha.extract.shared"}) {
t.Fatalf("unbinds = %#v, want alpha extract", got)
}
})
t.Run("specific binding restores broad unbind", func(t *testing.T) {
overrides, unbinds, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog,
[]cliReferenceRequest{{Selector: mustParseReferenceSelector(t, "alpha.extract.shared", "--reference"), Source: "alpha.txt"}},
[]cliReferenceUnbindRequest{{Selector: mustParseReferenceSelector(t, "shared", "--without-reference")}},
)
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
overrides, unbinds, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog, []cliReferenceRequest{ if got := referenceContractBindingLabels(overrides); !slices.Equal(got, []string{"alpha.extract.shared"}) {
{Selector: alphaShared, Source: "alpha-first.txt"}, t.Fatalf("overrides = %#v, want alpha extract", got)
{Selector: alphaShared, Source: "alpha-final.txt"},
{Selector: betaShared, Source: "beta-only.txt"},
}, nil)
if err != nil {
t.Fatal(err)
} }
if len(unbinds) != 0 { if got := referenceContractUnbindLabels(unbinds); slices.Contains(got, "alpha.extract.shared") || len(got) != 5 {
t.Fatalf("unbinds = %#v, want none", unbinds) t.Fatalf("unbinds = %#v, want all shared targets except alpha extract", got)
} }
effective, err := cfg.Resolve(config.ResolveInput{PipelineID: "demo", Catalog: catalog, ReferenceOverrides: overrides}) })
if err != nil {
t.Fatalf("resolve pipeline: %v", err) t.Run("same specificity conflicts", func(t *testing.T) {
} _, _, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog,
alpha := referenceContractLane(t, effective.ResolvedPipeline, "alpha") []cliReferenceRequest{{Selector: mustParseReferenceSelector(t, "alpha.shared", "--reference"), Source: "alpha.txt"}},
beta := referenceContractLane(t, effective.ResolvedPipeline, "beta") []cliReferenceUnbindRequest{{Selector: mustParseReferenceSelector(t, "alpha.shared", "--without-reference")}},
if source := referenceContractBindingSource(alpha.ExtractReferences.Bindings, "shared"); source != "alpha-final.txt" { )
t.Fatalf("alpha shared source = %q, want final exact-target override", source) if err == nil || !strings.Contains(err.Error(), "same specificity") {
} t.Fatalf("error = %v, want same-specificity conflict", err)
if source := referenceContractBindingSource(beta.ExtractReferences.Bindings, "shared"); source != "beta-only.txt" {
t.Fatalf("beta shared source = %q, want target-specific override", source)
} }
})
} }
func TestReferenceUnbindsRemoveOptionalAndProtectRequiredSlots(t *testing.T) { func TestReferenceUnbindsRemoveOptionalAndProtectRequiredSlots(t *testing.T) {
@@ -257,6 +329,52 @@ func TestReferenceMaterializationSeparatesCLIAndConfigPathOrigins(t *testing.T)
} }
} }
func TestPipelineScopedCLIReferenceProtectsGeneratedHandoff(t *testing.T) {
cfg := referenceContractConfig()
profile := cfg.Pipelines["demo"]
alpha := profile.Artifacts["alpha"]
beta := profile.Artifacts["beta"]
alpha.Extract.References["shared"] = pipeline.GeneratedReference("produce", "beta")
profile.Artifacts = nil
profile.Steps = []pipeline.PipelineStepProfile{
{ID: "produce", Artifacts: map[string]pipeline.ArtifactLaneProfile{"beta": beta}},
{ID: "consume", Artifacts: map[string]pipeline.ArtifactLaneProfile{"alpha": alpha}},
}
cfg.Pipelines["demo"] = profile
catalog := referenceContractCatalog(t, true, true)
t.Run("binding conflicts before file access", func(t *testing.T) {
overrides, _, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog, []cliReferenceRequest{{
Selector: mustParseReferenceSelector(t, "shared", "--reference"),
Source: filepath.Join(t.TempDir(), "never-read.json"),
}}, nil)
if err != nil {
t.Fatalf("expand CLI reference: %v", err)
}
_, err = cfg.Resolve(config.ResolveInput{PipelineID: "demo", Catalog: catalog, ReferenceOverrides: overrides})
if err == nil || !strings.Contains(err.Error(), "conflicting generated and external bindings") || strings.Contains(err.Error(), "never-read.json") {
t.Fatalf("resolve error = %v, want generated/external conflict before file access", err)
}
})
t.Run("unbind leaves generated source intact", func(t *testing.T) {
_, unbinds, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog, nil, []cliReferenceUnbindRequest{{
Selector: mustParseReferenceSelector(t, "shared", "--without-reference"),
}})
if err != nil {
t.Fatalf("expand CLI unbind: %v", err)
}
effective, err := cfg.Resolve(config.ResolveInput{PipelineID: "demo", Catalog: catalog, ReferenceUnbinds: unbinds})
if err != nil {
t.Fatalf("resolve generated reference with CLI unbind: %v", err)
}
binding := referenceContractFindBinding(referenceContractLane(t, effective.ResolvedPipeline, "alpha").ExtractReferences.Bindings, "shared")
if binding == nil || binding.Artifact == nil || binding.Artifact.Step != "produce" || binding.Artifact.Lane != "beta" {
t.Fatalf("generated binding = %#v, want preserved produce/beta handoff", binding)
}
})
}
func TestReferenceTargetLookupUsesArtifactVariantsAndReportsMissingContext(t *testing.T) { func TestReferenceTargetLookupUsesArtifactVariantsAndReportsMissingContext(t *testing.T) {
cfg := referenceContractConfig() cfg := referenceContractConfig()
full := referenceContractCatalog(t, true, true) full := referenceContractCatalog(t, true, true)
@@ -368,7 +486,7 @@ func referenceContractCatalog(t *testing.T, includeBetaMerger, includeBetaNormal
register(registries.Chunkers.RegisterWithSpec(pipeline.ModuleSpec{Key: "reference/chunk", Stage: pipeline.StageChunk, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"source"}, Provides: []string{"chunks"}, ReferenceSlots: []contracts.ReferenceSlot{{Name: "chunk-slot"}, {Name: "required-chunk", Required: true}}}, func() (contracts.Chunker, error) { return stateTestChunker{}, nil })) register(registries.Chunkers.RegisterWithSpec(pipeline.ModuleSpec{Key: "reference/chunk", Stage: pipeline.StageChunk, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"source"}, Provides: []string{"chunks"}, ReferenceSlots: []contracts.ReferenceSlot{{Name: "chunk-slot"}, {Name: "required-chunk", Required: true}}}, func() (contracts.Chunker, error) { return stateTestChunker{}, nil }))
register(pipeline.RegisterArtifactCodec(registries.ArtifactCodecs, referenceContractCodecA{})) register(pipeline.RegisterArtifactCodec(registries.ArtifactCodecs, referenceContractCodecA{}))
register(pipeline.RegisterArtifactCodec(registries.ArtifactCodecs, referenceContractCodecB{})) register(pipeline.RegisterArtifactCodec(registries.ArtifactCodecs, referenceContractCodecB{}))
register(pipeline.RegisterExtractor(registries.Extractors, pipeline.ModuleSpec{Key: "reference/extract-alpha", Stage: pipeline.StageExtract, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: referenceContractKindAlpha, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "alpha-slot"}, {Name: "required-extract", Required: true}}}, func() (contracts.Extractor[stateTestArtifact], error) { return stateTestExtractor{}, nil })) register(pipeline.RegisterExtractor(registries.Extractors, pipeline.ModuleSpec{Key: "reference/extract-alpha", Stage: pipeline.StageExtract, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: referenceContractKindAlpha, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared", AcceptedArtifactKinds: []contracts.ArtifactKind{referenceContractKindBeta}, AcceptedMediaTypes: []string{"application/json"}}, {Name: "alpha-slot"}, {Name: "required-extract", Required: true}}}, func() (contracts.Extractor[stateTestArtifact], error) { return stateTestExtractor{}, nil }))
register(pipeline.RegisterExtractor(registries.Extractors, pipeline.ModuleSpec{Key: "reference/extract-beta", Stage: pipeline.StageExtract, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: referenceContractKindBeta, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "beta-slot"}, {Name: "required-extract", Required: true}}}, func() (contracts.Extractor[stateTestArtifact], error) { return stateTestExtractor{}, nil })) register(pipeline.RegisterExtractor(registries.Extractors, pipeline.ModuleSpec{Key: "reference/extract-beta", Stage: pipeline.StageExtract, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: referenceContractKindBeta, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "beta-slot"}, {Name: "required-extract", Required: true}}}, func() (contracts.Extractor[stateTestArtifact], error) { return stateTestExtractor{}, nil }))
register(pipeline.RegisterMerger(registries.Mergers, pipeline.ModuleSpec{Key: "reference/shared-merge", Stage: pipeline.StageMerge, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"artifact"}, Provides: []string{"merged"}, ArtifactKind: referenceContractKindAlpha, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "alpha-merge"}, {Name: "required-merge", Required: true}}}, func() (contracts.Merger[stateTestArtifact], error) { return stateTestMerger{}, nil })) register(pipeline.RegisterMerger(registries.Mergers, pipeline.ModuleSpec{Key: "reference/shared-merge", Stage: pipeline.StageMerge, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"artifact"}, Provides: []string{"merged"}, ArtifactKind: referenceContractKindAlpha, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "alpha-merge"}, {Name: "required-merge", Required: true}}}, func() (contracts.Merger[stateTestArtifact], error) { return stateTestMerger{}, nil }))
if includeBetaMerger { if includeBetaMerger {
@@ -444,6 +562,42 @@ func referenceContractBindingSource(bindings []pipeline.ReferenceBinding, slot s
return "" return ""
} }
func mustParseReferenceSelector(t *testing.T, value, flagName string) cliReferenceSelector {
t.Helper()
selector, err := parseReferenceSelector(value, flagName)
if err != nil {
t.Fatal(err)
}
return selector
}
func referenceContractBindingLabels(bindings []pipeline.ReferenceBinding) []string {
labels := make([]string, 0, len(bindings))
for _, binding := range bindings {
labels = append(labels, referenceContractBindingLabel(binding))
}
return labels
}
func referenceContractBindingLabel(binding pipeline.ReferenceBinding) string {
if binding.Stage == pipeline.StageChunk {
return "chunk." + binding.SlotName
}
return binding.LaneID + "." + string(binding.Stage) + "." + binding.SlotName
}
func referenceContractUnbindLabels(unbinds []pipeline.ReferenceUnbind) []string {
labels := make([]string, 0, len(unbinds))
for _, unbind := range unbinds {
labels = append(labels, referenceContractBindingLabel(pipeline.ReferenceBinding{
Stage: unbind.Stage,
LaneID: unbind.LaneID,
SlotName: unbind.SlotName,
}))
}
return labels
}
func referenceContractFindBinding(bindings []pipeline.ReferenceBinding, slot string) *pipeline.ReferenceBinding { func referenceContractFindBinding(bindings []pipeline.ReferenceBinding, slot string) *pipeline.ReferenceBinding {
for i := range bindings { for i := range bindings {
if bindings[i].SlotName == slot { if bindings[i].SlotName == slot {

View File

@@ -177,7 +177,7 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
fs.Var(&llmProfile, "llm-profile", "LLM profile override") fs.Var(&llmProfile, "llm-profile", "LLM profile override")
fs.Var(&reasoningEffort, "reasoning-effort", "reasoning effort override") fs.Var(&reasoningEffort, "reasoning-effort", "reasoning effort override")
fs.Var(&chunkCache, "chunk_cache", "chunk plan cache mode: auto, bypass, or refresh") fs.Var(&chunkCache, "chunk_cache", "chunk plan cache mode: auto, bypass, or refresh")
fs.Var(&referenceFlags, "reference", "reference binding, as slot=path, chunk.slot=path, merge.slot=path, lane.slot=path, lane.extract.slot=path, lane.merge.slot=path, or lane.normalize.slot=path") fs.Var(&referenceFlags, "reference", "reference binding, as slot=path, chunk.slot=path, lane.slot=path, lane.extract.slot=path, lane.merge.slot=path, or lane.normalize.slot=path")
fs.Var(&withoutReferenceFlags, "without-reference", "unbind a reference, using the same selector forms as --reference") fs.Var(&withoutReferenceFlags, "without-reference", "unbind a reference, using the same selector forms as --reference")
fs.Var(&recomputeStep, "recompute-step", "recompute one ordered pipeline step and dependent lanes") fs.Var(&recomputeStep, "recompute-step", "recompute one ordered pipeline step and dependent lanes")
if err := validateRunFlagValues(args); err != nil { if err := validateRunFlagValues(args); err != nil {
@@ -375,7 +375,7 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
if err != nil { if err != nil {
return failPipelineCommand(stderr, commandState, terminalWriter, fmt.Errorf("resolve working directory: %w", err)) return failPipelineCommand(stderr, commandState, terminalWriter, fmt.Errorf("resolve working directory: %w", err))
} }
materialized, referenceWarnings, err := pipeline.MaterializeReferences(effective.ResolvedPipeline, catalog, pipeline.ReferenceMaterializationOptions{ materialized, referenceDiagnostics, err := pipeline.MaterializeReferences(effective.ResolvedPipeline, catalog, pipeline.ReferenceMaterializationOptions{
ConfigPath: loadedConfigPath, ConfigPath: loadedConfigPath,
WorkingDir: workingDir, WorkingDir: workingDir,
}) })
@@ -462,7 +462,7 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
StartedAt: startedAt, StartedAt: startedAt,
LLMProfiles: llmProfiles, LLMProfiles: llmProfiles,
Metadata: runMetadata(effective.Config.Output.Directory, debugPath), Metadata: runMetadata(effective.Config.Output.Directory, debugPath),
Warnings: referenceWarnings, Diagnostics: referenceDiagnostics,
ChunkCacheMode: effective.Config.Cache.ChunkPlans.Mode, ChunkCacheMode: effective.Config.Cache.ChunkPlans.Mode,
ChunkPlans: chunkPlans, ChunkPlans: chunkPlans,
Checkpoints: checkpointRecorder, Checkpoints: checkpointRecorder,
@@ -471,7 +471,10 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
Debug: debugRecorder, Debug: debugRecorder,
ExtractWorkers: cfg.Concurrency.StageWorkers["extract"], ExtractWorkers: cfg.Concurrency.StageWorkers["extract"],
}) })
commandState.observeOutput(output) diagnosticProjection, diagnosticErr := contracts.ProjectDiagnosticCollection(output.Diagnostics)
if diagnosticErr == nil {
commandState.observeOutput(output, diagnosticProjection)
}
if err != nil { if err != nil {
primaryErr := fmt.Errorf("run pipeline %q: %w", pipelineID, err) primaryErr := fmt.Errorf("run pipeline %q: %w", pipelineID, err)
if output.Manifest.PipelineID != "" { if output.Manifest.PipelineID != "" {
@@ -479,15 +482,21 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
return failPipelineCommand(stderr, commandState, terminalWriter, primaryErr, fmt.Errorf("write debug summary: %w", summaryErr)) return failPipelineCommand(stderr, commandState, terminalWriter, primaryErr, fmt.Errorf("write debug summary: %w", summaryErr))
} }
} }
if diagnosticErr != nil {
return failPipelineCommand(stderr, commandState, terminalWriter, primaryErr, fmt.Errorf("summarize run diagnostics: %w", diagnosticErr))
}
return failPipelineCommand(stderr, commandState, terminalWriter, primaryErr) return failPipelineCommand(stderr, commandState, terminalWriter, primaryErr)
} }
if diagnosticErr != nil {
return failPipelineCommand(stderr, commandState, terminalWriter, fmt.Errorf("summarize run diagnostics: %w", diagnosticErr))
}
if err := writePartialSummary(summary, output); err != nil { if err := writePartialSummary(summary, output); err != nil {
return failPipelineCommand(stderr, commandState, terminalWriter, fmt.Errorf("write debug summary: %w", err)) return failPipelineCommand(stderr, commandState, terminalWriter, fmt.Errorf("write debug summary: %w", err))
} }
var encodedResult []byte var encodedResult []byte
if *machineOutput { if *machineOutput {
result, err := newRunResult(effective.ResolvedPipeline, output, runOutputDir, debugPath) result, err := newRunResultWithDiagnostics(effective.ResolvedPipeline, output, runOutputDir, debugPath, diagnosticProjection)
if err != nil { if err != nil {
return failPipelineCommand(stderr, commandState, terminalWriter, err) return failPipelineCommand(stderr, commandState, terminalWriter, err)
} }
@@ -513,12 +522,25 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
fmt.Fprintf(stdout, "debug=%s\n", debugPath) fmt.Fprintf(stdout, "debug=%s\n", debugPath)
} }
} }
if len(output.Warnings) > 0 { if warningGroups := len(diagnosticProjection.Warnings); warningGroups > 0 {
fmt.Fprintf(stderr, "notarius: run completed with %d warning(s)\n", len(output.Warnings)) fmt.Fprintf(stderr, "notarius: run completed with %d warning group(s), %d occurrence(s)", warningGroups, diagnosticProjection.WarningOccurrenceCount)
if warningFile, ok := logicalOutputFile(output.OutputFiles, "warnings.json"); ok {
fmt.Fprintf(stderr, "; details=%s", filepath.Join(runOutputDir, warningFile))
}
fmt.Fprintln(stderr)
} }
return 0 return 0
} }
func logicalOutputFile(files []contracts.OutputFile, name string) (string, bool) {
for _, file := range files {
if file.Name == name {
return file.Name, true
}
}
return "", false
}
func writeSummary(summary *debugbundle.SummaryWriter, write func() error) error { func writeSummary(summary *debugbundle.SummaryWriter, write func() error) error {
if summary == nil { if summary == nil {
return nil return nil
@@ -538,7 +560,7 @@ func writePartialSummary(summary *debugbundle.SummaryWriter, output pipeline.Run
return err return err
} }
} }
if err := summary.WriteWarnings(output.Warnings); err != nil { if err := summary.WriteDiagnostics(output.Diagnostics); err != nil {
return err return err
} }
return summary.WriteCheckpointEvents(output.CheckpointEvents) return summary.WriteCheckpointEvents(output.CheckpointEvents)
@@ -1265,11 +1287,21 @@ type cliReferenceUnbindRequest struct {
} }
type cliReferenceSelector struct { type cliReferenceSelector struct {
Scope cliReferenceSelectorScope
LaneID string LaneID string
Stage pipeline.ModuleStage Stage pipeline.ModuleStage
SlotName string SlotName string
} }
type cliReferenceSelectorScope uint8
const (
cliReferenceScopePipeline cliReferenceSelectorScope = iota
cliReferenceScopeLane
cliReferenceScopeChunk
cliReferenceScopeBinding
)
func parseReferenceFlags(values []string) ([]cliReferenceRequest, error) { func parseReferenceFlags(values []string) ([]cliReferenceRequest, error) {
if len(values) == 0 { if len(values) == 0 {
return nil, nil return nil, nil
@@ -1278,7 +1310,7 @@ func parseReferenceFlags(values []string) ([]cliReferenceRequest, error) {
for _, raw := range values { for _, raw := range values {
name, source, ok := strings.Cut(raw, "=") name, source, ok := strings.Cut(raw, "=")
if !ok { if !ok {
return nil, fmt.Errorf("--reference must use slot=path or lane.slot=path") return nil, fmt.Errorf("--reference must use slot=path, lane.slot=path, or lane.stage.slot=path")
} }
if strings.TrimSpace(source) == "" { if strings.TrimSpace(source) == "" {
return nil, fmt.Errorf("--reference path must not be empty; use --without-reference to unbind") return nil, fmt.Errorf("--reference path must not be empty; use --without-reference to unbind")
@@ -1328,17 +1360,14 @@ func parseReferenceSelector(raw string, flagName string) (cliReferenceSelector,
} }
switch len(parts) { switch len(parts) {
case 1: case 1:
return cliReferenceSelector{SlotName: strings.TrimSpace(parts[0])}, nil return cliReferenceSelector{Scope: cliReferenceScopePipeline, SlotName: strings.TrimSpace(parts[0])}, nil
case 2: case 2:
first := strings.TrimSpace(parts[0]) first := strings.TrimSpace(parts[0])
slotName := strings.TrimSpace(parts[1]) slotName := strings.TrimSpace(parts[1])
if first == string(pipeline.StageChunk) { if first == string(pipeline.StageChunk) {
return cliReferenceSelector{Stage: pipeline.StageChunk, SlotName: slotName}, nil return cliReferenceSelector{Scope: cliReferenceScopeChunk, Stage: pipeline.StageChunk, SlotName: slotName}, nil
} }
if first == string(pipeline.StageMerge) { return cliReferenceSelector{Scope: cliReferenceScopeLane, LaneID: first, SlotName: slotName}, nil
return cliReferenceSelector{Stage: pipeline.StageMerge, SlotName: slotName}, nil
}
return cliReferenceSelector{LaneID: first, SlotName: slotName}, nil
case 3: case 3:
laneID := strings.TrimSpace(parts[0]) laneID := strings.TrimSpace(parts[0])
stage := pipeline.ModuleStage(strings.TrimSpace(parts[1])) stage := pipeline.ModuleStage(strings.TrimSpace(parts[1]))
@@ -1346,9 +1375,9 @@ func parseReferenceSelector(raw string, flagName string) (cliReferenceSelector,
if stage != pipeline.StageExtract && stage != pipeline.StageMerge && stage != pipeline.StageNormalize { if stage != pipeline.StageExtract && stage != pipeline.StageMerge && stage != pipeline.StageNormalize {
return cliReferenceSelector{}, fmt.Errorf("%s lane-qualified selector must use lane.extract.slot, lane.merge.slot, or lane.normalize.slot", flagName) return cliReferenceSelector{}, fmt.Errorf("%s lane-qualified selector must use lane.extract.slot, lane.merge.slot, or lane.normalize.slot", flagName)
} }
return cliReferenceSelector{LaneID: laneID, Stage: stage, SlotName: slotName}, nil return cliReferenceSelector{Scope: cliReferenceScopeBinding, LaneID: laneID, Stage: stage, SlotName: slotName}, nil
default: default:
return cliReferenceSelector{}, fmt.Errorf("%s must use slot, chunk.slot, merge.slot, lane.slot, lane.extract.slot, lane.merge.slot, or lane.normalize.slot", flagName) return cliReferenceSelector{}, fmt.Errorf("%s must use slot, chunk.slot, lane.slot, lane.extract.slot, lane.merge.slot, or lane.normalize.slot", flagName)
} }
} }
@@ -1369,37 +1398,153 @@ func resolveCLIReferenceRequests(
return nil, nil, err return nil, nil, err
} }
overrides := make([]pipeline.ReferenceBinding, 0, len(referenceRequests)) // Broad CLI selectors are only presentation syntax. Collapse them into one
// highest-specificity action per concrete framework target before pipeline
// resolution so the generic reference contract stays stage-and-lane exact.
actions := make(map[cliReferenceTargetKey]resolvedCLIReferenceAction)
for _, request := range referenceRequests { for _, request := range referenceRequests {
target, err := resolveCLIReferenceTarget(targets, request.Selector) matches, err := resolveCLIReferenceTargets(targets, request.Selector)
if err != nil { if err != nil {
return nil, nil, err return nil, nil, err
} }
for _, target := range matches {
candidate := resolvedCLIReferenceAction{
kind: cliReferenceActionBind,
selector: request.Selector,
target: target,
slotName: request.Selector.SlotName,
source: request.Source,
specificity: request.Selector.specificity(),
}
if err := mergeCLIReferenceAction(actions, candidate); err != nil {
return nil, nil, err
}
}
}
for _, request := range unbindRequests {
matches, err := resolveCLIReferenceTargets(targets, request.Selector)
if err != nil {
return nil, nil, err
}
for _, target := range matches {
candidate := resolvedCLIReferenceAction{
kind: cliReferenceActionUnbind,
selector: request.Selector,
target: target,
slotName: request.Selector.SlotName,
specificity: request.Selector.specificity(),
}
if err := mergeCLIReferenceAction(actions, candidate); err != nil {
return nil, nil, err
}
}
}
resolved := make([]resolvedCLIReferenceAction, 0, len(actions))
for _, action := range actions {
resolved = append(resolved, action)
}
sort.Slice(resolved, func(i, j int) bool {
left, right := resolved[i], resolved[j]
if left.target.laneID != right.target.laneID {
return left.target.laneID < right.target.laneID
}
if left.target.stage != right.target.stage {
return referenceStageOrder(left.target.stage) < referenceStageOrder(right.target.stage)
}
return left.slotName < right.slotName
})
overrides := make([]pipeline.ReferenceBinding, 0, len(resolved))
unbinds := make([]pipeline.ReferenceUnbind, 0, len(resolved))
for _, action := range resolved {
switch action.kind {
case cliReferenceActionBind:
overrides = append(overrides, pipeline.ReferenceBinding{ overrides = append(overrides, pipeline.ReferenceBinding{
Stage: target.stage, Stage: action.target.stage,
LaneID: target.laneID, LaneID: action.target.laneID,
SlotName: request.Selector.SlotName, SlotName: action.slotName,
Source: request.Source, Source: action.source,
BindingSource: contracts.ReferenceBindingSourceCLI, BindingSource: contracts.ReferenceBindingSourceCLI,
}) })
} case cliReferenceActionUnbind:
unbinds := make([]pipeline.ReferenceUnbind, 0, len(unbindRequests))
for _, request := range unbindRequests {
target, err := resolveCLIReferenceTarget(targets, request.Selector)
if err != nil {
return nil, nil, err
}
unbinds = append(unbinds, pipeline.ReferenceUnbind{ unbinds = append(unbinds, pipeline.ReferenceUnbind{
Stage: target.stage, Stage: action.target.stage,
LaneID: target.laneID, LaneID: action.target.laneID,
SlotName: request.Selector.SlotName, SlotName: action.slotName,
}) })
} }
}
return overrides, unbinds, nil return overrides, unbinds, nil
} }
type cliReferenceActionKind uint8
const (
cliReferenceActionBind cliReferenceActionKind = iota
cliReferenceActionUnbind
)
type cliReferenceTargetKey struct {
stage pipeline.ModuleStage
laneID string
slotName string
}
type resolvedCLIReferenceAction struct {
kind cliReferenceActionKind
selector cliReferenceSelector
target selectedReferenceTarget
slotName string
source string
specificity int
}
func mergeCLIReferenceAction(actions map[cliReferenceTargetKey]resolvedCLIReferenceAction, candidate resolvedCLIReferenceAction) error {
key := cliReferenceTargetKey{stage: candidate.target.stage, laneID: candidate.target.laneID, slotName: candidate.slotName}
current, ok := actions[key]
if !ok || candidate.specificity > current.specificity {
actions[key] = candidate
return nil
}
if candidate.specificity < current.specificity {
return nil
}
if candidate.kind != current.kind {
return fmt.Errorf("reference target %q slot %q is both bound by %q and unbound by %q at the same specificity", targetLabel(candidate.target), candidate.slotName, current.selector.String(), candidate.selector.String())
}
actions[key] = candidate
return nil
}
func (selector cliReferenceSelector) specificity() int {
switch selector.Scope {
case cliReferenceScopePipeline:
return 0
case cliReferenceScopeLane:
return 1
case cliReferenceScopeChunk, cliReferenceScopeBinding:
return 2
default:
return -1
}
}
func (selector cliReferenceSelector) String() string {
switch selector.Scope {
case cliReferenceScopePipeline:
return selector.SlotName
case cliReferenceScopeLane:
return selector.LaneID + "." + selector.SlotName
case cliReferenceScopeChunk:
return "chunk." + selector.SlotName
case cliReferenceScopeBinding:
return selector.LaneID + "." + string(selector.Stage) + "." + selector.SlotName
default:
return selector.SlotName
}
}
type selectedReferenceTarget struct { type selectedReferenceTarget struct {
laneID string laneID string
stage pipeline.ModuleStage stage pipeline.ModuleStage
@@ -1616,68 +1761,39 @@ func referenceSlotSet(slots []contracts.ReferenceSlot) map[string]struct{} {
return slotSet return slotSet
} }
func resolveCLIReferenceTarget(targets []selectedReferenceTarget, selector cliReferenceSelector) (selectedReferenceTarget, error) { func resolveCLIReferenceTargets(targets []selectedReferenceTarget, selector cliReferenceSelector) ([]selectedReferenceTarget, error) {
slotName := strings.TrimSpace(selector.SlotName) slotName := strings.TrimSpace(selector.SlotName)
if slotName == "" { if slotName == "" {
return selectedReferenceTarget{}, fmt.Errorf("reference slot must not be empty") return nil, fmt.Errorf("reference slot must not be empty")
} }
if selector.Stage == pipeline.StageChunk { switch selector.Scope {
case cliReferenceScopePipeline:
matches := make([]selectedReferenceTarget, 0, len(targets))
for _, target := range targets {
if _, ok := target.slots[slotName]; ok {
matches = append(matches, target)
}
}
if len(matches) == 0 {
return nil, fmt.Errorf("reference slot %q is not declared by any selected target", slotName)
}
return matches, nil
case cliReferenceScopeChunk:
for _, target := range targets { for _, target := range targets {
if target.stage != pipeline.StageChunk { if target.stage != pipeline.StageChunk {
continue continue
} }
if _, ok := target.slots[slotName]; !ok { if _, ok := target.slots[slotName]; !ok {
return selectedReferenceTarget{}, fmt.Errorf("reference slot %q is not declared by chunk module %q", slotName, target.module) return nil, fmt.Errorf("reference slot %q is not declared by chunk module %q", slotName, target.module)
} }
return target, nil return []selectedReferenceTarget{target}, nil
} }
return selectedReferenceTarget{}, fmt.Errorf("reference chunk target is not selected") return nil, fmt.Errorf("reference chunk target is not selected")
} case cliReferenceScopeLane:
if selector.Stage == pipeline.StageExtract || selector.Stage == pipeline.StageMerge || selector.Stage == pipeline.StageNormalize {
if selector.LaneID == "" && selector.Stage == pipeline.StageMerge {
return resolveCLIReferenceStageTarget(targets, selector.Stage, slotName)
}
for _, target := range targets {
if target.laneID == selector.LaneID && target.stage == selector.Stage {
if _, ok := target.slots[slotName]; !ok {
return selectedReferenceTarget{}, fmt.Errorf("reference slot %q is not declared by selected %s target %q", slotName, selector.Stage, targetLabel(target))
}
return target, nil
}
}
return selectedReferenceTarget{}, fmt.Errorf("reference lane %q is not selected", selector.LaneID)
}
if strings.TrimSpace(selector.LaneID) != "" {
return resolveCLIReferenceLaneTarget(targets, strings.TrimSpace(selector.LaneID), slotName)
}
return resolveCLIReferenceFlatTarget(targets, slotName)
}
func resolveCLIReferenceStageTarget(targets []selectedReferenceTarget, stage pipeline.ModuleStage, slotName string) (selectedReferenceTarget, error) {
matches := make([]selectedReferenceTarget, 0, 2)
for _, target := range targets {
if target.stage != stage {
continue
}
if _, ok := target.slots[slotName]; ok {
matches = append(matches, target)
}
}
switch len(matches) {
case 0:
return selectedReferenceTarget{}, fmt.Errorf("reference slot %q is not declared by any selected %s target", slotName, stage)
case 1:
return matches[0], nil
default:
return selectedReferenceTarget{}, fmt.Errorf("reference slot %q is declared by multiple selected %s targets (%s); use a more specific selector such as %s", slotName, stage, targetList(matches), selectorSuggestions(matches, slotName))
}
}
func resolveCLIReferenceLaneTarget(targets []selectedReferenceTarget, laneID string, slotName string) (selectedReferenceTarget, error) {
laneSelected := false laneSelected := false
matches := make([]selectedReferenceTarget, 0, 2) matches := make([]selectedReferenceTarget, 0, 3)
for _, target := range targets { for _, target := range targets {
if target.laneID != laneID { if target.laneID != selector.LaneID {
continue continue
} }
laneSelected = true laneSelected = true
@@ -1686,44 +1802,36 @@ func resolveCLIReferenceLaneTarget(targets []selectedReferenceTarget, laneID str
} }
} }
if !laneSelected { if !laneSelected {
return selectedReferenceTarget{}, fmt.Errorf("reference lane %q is not selected", laneID) return nil, fmt.Errorf("reference lane %q is not selected", selector.LaneID)
} }
switch len(matches) { if len(matches) == 0 {
case 0: return nil, fmt.Errorf("reference slot %q is not declared by selected lane %q", slotName, selector.LaneID)
return selectedReferenceTarget{}, fmt.Errorf("reference slot %q is not declared by selected lane %q", slotName, laneID)
case 1:
return matches[0], nil
default:
return selectedReferenceTarget{}, fmt.Errorf("reference slot %q is declared by multiple selected targets in lane %q (%s); use a more specific selector such as %s", slotName, laneID, targetList(matches), selectorSuggestions(matches, slotName))
} }
} return matches, nil
case cliReferenceScopeBinding:
func resolveCLIReferenceFlatTarget(targets []selectedReferenceTarget, slotName string) (selectedReferenceTarget, error) { laneSelected := false
matches := make([]selectedReferenceTarget, 0, 2)
for _, target := range targets { for _, target := range targets {
if _, ok := target.slots[slotName]; ok { if target.laneID != selector.LaneID {
matches = append(matches, target) continue
} }
laneSelected = true
if target.stage != selector.Stage {
continue
} }
switch len(matches) { if _, ok := target.slots[slotName]; !ok {
case 0: return nil, fmt.Errorf("reference slot %q is not declared by selected %s target %q", slotName, selector.Stage, targetLabel(target))
return selectedReferenceTarget{}, fmt.Errorf("reference slot %q is not declared by any selected reference target", slotName) }
case 1: return []selectedReferenceTarget{target}, nil
return matches[0], nil }
if !laneSelected {
return nil, fmt.Errorf("reference lane %q is not selected", selector.LaneID)
}
return nil, fmt.Errorf("reference %s target is not selected for lane %q", selector.Stage, selector.LaneID)
default: default:
return selectedReferenceTarget{}, fmt.Errorf("reference slot %q is declared by multiple selected targets (%s); use a more specific selector such as %s", slotName, targetList(matches), selectorSuggestions(matches, slotName)) return nil, fmt.Errorf("reference selector has unknown scope")
} }
} }
func targetList(targets []selectedReferenceTarget) string {
labels := make([]string, 0, len(targets))
for _, target := range targets {
labels = append(labels, targetLabel(target))
}
sort.Strings(labels)
return strings.Join(labels, ", ")
}
func targetLabel(target selectedReferenceTarget) string { func targetLabel(target selectedReferenceTarget) string {
if target.stage == pipeline.StageChunk { if target.stage == pipeline.StageChunk {
return "chunk" return "chunk"
@@ -1731,17 +1839,19 @@ func targetLabel(target selectedReferenceTarget) string {
return target.laneID + "." + string(target.stage) return target.laneID + "." + string(target.stage)
} }
func selectorSuggestions(targets []selectedReferenceTarget, slotName string) string { func referenceStageOrder(stage pipeline.ModuleStage) int {
suggestions := make([]string, 0, len(targets)) switch stage {
for _, target := range targets { case pipeline.StageChunk:
if target.stage == pipeline.StageChunk { return 0
suggestions = append(suggestions, "chunk."+slotName) case pipeline.StageExtract:
continue return 1
case pipeline.StageMerge:
return 2
case pipeline.StageNormalize:
return 3
default:
return 4
} }
suggestions = append(suggestions, target.laneID+"."+string(target.stage)+"."+slotName)
}
sort.Strings(suggestions)
return strings.Join(suggestions, " or ")
} }
func sortedPipelineIDs(cfg config.Config) []string { func sortedPipelineIDs(cfg config.Config) []string {

View File

@@ -590,10 +590,11 @@ func TestRunWarningsRemainSuccessfulAndReachDurableSurfaces(t *testing.T) {
roots := newStateTestRoots(t) roots := newStateTestRoots(t)
harness := newStateTestHarness() harness := newStateTestHarness()
harness.includeWarnings = true harness.includeWarnings = true
harness.chunkWarnings = []contracts.Warning{{Scope: "chunk", ReasonCode: "contract-warning", Message: "warning retained"}} harness.includeWarningFile = true
harness.chunkDiagnostics = []contracts.ProducerDiagnostic{stateTestDiagnostic("chunk", "contract-warning", "warning retained")}
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass", "--debug"}, &stdout, &stderr, harness.options()) code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass", "--debug"}, &stdout, &stderr, harness.options())
if code != 0 || !strings.Contains(stdout.String(), "outputs=1") || !strings.Contains(stderr.String(), "1 warning(s)") { if code != 0 || !strings.Contains(stdout.String(), "outputs=1") || !strings.Contains(stderr.String(), "1 warning group(s), 1 occurrence(s)") || !strings.Contains(stderr.String(), "warnings.json") {
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String()) t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
} }
outputPath := filepath.Join(onlyChildDir(t, roots.output), "result.json") outputPath := filepath.Join(onlyChildDir(t, roots.output), "result.json")
@@ -601,11 +602,12 @@ func TestRunWarningsRemainSuccessfulAndReachDurableSurfaces(t *testing.T) {
if err != nil || !strings.Contains(string(output), "contract-warning") { if err != nil || !strings.Contains(string(output), "contract-warning") {
t.Fatalf("durable output = %q, %v", output, err) t.Fatalf("durable output = %q, %v", output, err)
} }
assertFile(t, filepath.Join(filepath.Dir(outputPath), "warnings.json"))
bundle := onlyChildDir(t, roots.debug) bundle := onlyChildDir(t, roots.debug)
var warnings []contracts.Warning var diagnostics contracts.DiagnosticCollection
readStateTestSummaryJSON(t, bundle, "warnings.json", &warnings) readStateTestSummaryJSON(t, bundle, "final-diagnostics.json", &diagnostics)
if len(warnings) != 1 || warnings[0].ReasonCode != "contract-warning" { if len(diagnostics.Groups) != 1 || diagnostics.Groups[0].ReasonCode != "contract-warning" {
t.Fatalf("debug warnings = %#v", warnings) t.Fatalf("debug diagnostics = %#v", diagnostics)
} }
} }

View File

@@ -8,10 +8,11 @@ import (
"strings" "strings"
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts" "gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline" "gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
) )
const runResultSchemaVersion = "notarius.run-result.v1" const runResultSchemaVersion = "notarius.run-result.v2"
type runResult struct { type runResult struct {
SchemaVersion string `json:"schema_version"` SchemaVersion string `json:"schema_version"`
@@ -21,13 +22,25 @@ type runResult struct {
IndexFile string `json:"index_file,omitempty"` IndexFile string `json:"index_file,omitempty"`
NormalizedOutputCount int `json:"normalized_output_count"` NormalizedOutputCount int `json:"normalized_output_count"`
RejectedOutputCount int `json:"rejected_output_count"` RejectedOutputCount int `json:"rejected_output_count"`
WarningCount int `json:"warning_count"` WarningGroupCount int `json:"warning_group_count"`
WarningOccurrenceCount int `json:"warning_occurrence_count"`
DiagnosticGroupCount int `json:"diagnostic_group_count"`
DiagnosticOccurrenceCount int `json:"diagnostic_occurrence_count"`
DiagnosticsTruncated bool `json:"diagnostics_truncated"`
ValidationStatus string `json:"validation_status"` ValidationStatus string `json:"validation_status"`
ValidationSummaries []artifacts.ValidationSummary `json:"validation_summaries,omitempty"` ValidationSummaries []artifacts.ValidationSummary `json:"validation_summaries,omitempty"`
DebugDirectory string `json:"debug_directory,omitempty"` DebugDirectory string `json:"debug_directory,omitempty"`
} }
func newRunResult(resolved pipeline.ResolvedPipeline, output pipeline.RunOutput, outputDirectory, debugDirectory string) (runResult, error) { func newRunResult(resolved pipeline.ResolvedPipeline, output pipeline.RunOutput, outputDirectory, debugDirectory string) (runResult, error) {
diagnosticProjection, err := contracts.ProjectDiagnosticCollection(output.Diagnostics)
if err != nil {
return runResult{}, fmt.Errorf("summarize run diagnostics: %w", err)
}
return newRunResultWithDiagnostics(resolved, output, outputDirectory, debugDirectory, diagnosticProjection)
}
func newRunResultWithDiagnostics(resolved pipeline.ResolvedPipeline, output pipeline.RunOutput, outputDirectory, debugDirectory string, diagnosticProjection contracts.DiagnosticProjection) (runResult, error) {
if strings.TrimSpace(output.Manifest.RunID) == "" { if strings.TrimSpace(output.Manifest.RunID) == "" {
return runResult{}, fmt.Errorf("run result requires a run ID") return runResult{}, fmt.Errorf("run result requires a run ID")
} }
@@ -59,7 +72,11 @@ func newRunResult(resolved pipeline.ResolvedPipeline, output pipeline.RunOutput,
OutputDirectory: absOutputDirectory, OutputDirectory: absOutputDirectory,
NormalizedOutputCount: len(output.NormalizeOutputs), NormalizedOutputCount: len(output.NormalizeOutputs),
RejectedOutputCount: len(output.Rejected), RejectedOutputCount: len(output.Rejected),
WarningCount: len(output.Warnings), WarningGroupCount: len(diagnosticProjection.Warnings),
WarningOccurrenceCount: diagnosticProjection.WarningOccurrenceCount,
DiagnosticGroupCount: len(diagnosticProjection.Diagnostics),
DiagnosticOccurrenceCount: diagnosticProjection.DiagnosticOccurrenceCount,
DiagnosticsTruncated: output.Diagnostics.Truncated,
ValidationStatus: output.Manifest.ValidationStatus, ValidationStatus: output.Manifest.ValidationStatus,
ValidationSummaries: cloneValidationSummaries(output.Manifest.ValidationSummaries), ValidationSummaries: cloneValidationSummaries(output.Manifest.ValidationSummaries),
} }

View File

@@ -27,7 +27,7 @@ func TestMaintainedMinimalInvocationEmitsRunResult(t *testing.T) {
} }
receipt := decodeRunResultDocument(t, stdout.String()) receipt := decodeRunResultDocument(t, stdout.String())
if got := receipt["schema_version"]; got != "notarius.run-result.v1" { if got := receipt["schema_version"]; got != "notarius.run-result.v2" {
t.Fatalf("schema_version = %q", got) t.Fatalf("schema_version = %q", got)
} }
if got := receipt["run_id"]; got != productionRunID { if got := receipt["run_id"]; got != productionRunID {
@@ -45,8 +45,8 @@ func TestMaintainedMinimalInvocationEmitsRunResult(t *testing.T) {
if got := receipt["rejected_output_count"]; got != float64(0) { if got := receipt["rejected_output_count"]; got != float64(0) {
t.Fatalf("rejected_output_count = %v", got) t.Fatalf("rejected_output_count = %v", got)
} }
if got := receipt["warning_count"]; got != float64(0) { if got := receipt["warning_group_count"]; got != float64(0) || receipt["warning_occurrence_count"] != float64(0) || receipt["diagnostic_group_count"] != float64(0) || receipt["diagnostic_occurrence_count"] != float64(0) || receipt["diagnostics_truncated"] != false {
t.Fatalf("warning_count = %v", got) t.Fatalf("diagnostic counts = %#v", receipt)
} }
if got := receipt["validation_status"]; got != "approved" { if got := receipt["validation_status"]; got != "approved" {
t.Fatalf("validation_status = %q", got) t.Fatalf("validation_status = %q", got)
@@ -63,19 +63,19 @@ func TestMaintainedMinimalInvocationEmitsRunResult(t *testing.T) {
func TestRunResultReportsWarningsAndDebugBundle(t *testing.T) { func TestRunResultReportsWarningsAndDebugBundle(t *testing.T) {
roots := newStateTestRoots(t) roots := newStateTestRoots(t)
harness := newStateTestHarness() harness := newStateTestHarness()
harness.chunkWarnings = []contracts.Warning{{Scope: "chunk", ReasonCode: "contract-warning", Message: "warning retained"}} harness.chunkDiagnostics = []contracts.ProducerDiagnostic{stateTestDiagnostic("chunk", "contract-warning", "warning retained")}
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
code := RunWithOptions([]string{ code := RunWithOptions([]string{
"run", "sample", "--config", roots.config, "--input", roots.input, "run", "sample", "--config", roots.config, "--input", roots.input,
"--chunk_cache", "bypass", "--debug", "--json", "--chunk_cache", "bypass", "--debug", "--json",
}, &stdout, &stderr, harness.options()) }, &stdout, &stderr, harness.options())
if code != 0 || !strings.Contains(stderr.String(), "1 warning(s)") { if code != 0 || !strings.Contains(stderr.String(), "1 warning group(s), 1 occurrence(s)") || strings.Contains(stderr.String(), "warnings.json") {
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String()) t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
} }
receipt := decodeRunResultDocument(t, stdout.String()) receipt := decodeRunResultDocument(t, stdout.String())
if got := receipt["warning_count"]; got != float64(1) { if got := receipt["warning_group_count"]; got != float64(1) || receipt["warning_occurrence_count"] != float64(1) || receipt["diagnostic_group_count"] != float64(0) || receipt["diagnostic_occurrence_count"] != float64(0) || receipt["diagnostics_truncated"] != false {
t.Fatalf("warning_count = %v", got) t.Fatalf("diagnostic counts = %#v", receipt)
} }
debugDirectory, ok := receipt["debug_directory"].(string) debugDirectory, ok := receipt["debug_directory"].(string)
if !ok || !filepath.IsAbs(debugDirectory) || debugDirectory != onlyChildDir(t, roots.debug) { if !ok || !filepath.IsAbs(debugDirectory) || debugDirectory != onlyChildDir(t, roots.debug) {

View File

@@ -52,8 +52,8 @@ func TestRunResultEncodesRequiredFieldsAndCounts(t *testing.T) {
if got := decoded["rejected_output_count"]; got != float64(1) { if got := decoded["rejected_output_count"]; got != float64(1) {
t.Fatalf("rejected_output_count = %v", got) t.Fatalf("rejected_output_count = %v", got)
} }
if got := decoded["warning_count"]; got != float64(1) { if got := decoded["warning_group_count"]; got != float64(1) || decoded["warning_occurrence_count"] != float64(1) || decoded["diagnostic_group_count"] != float64(0) || decoded["diagnostic_occurrence_count"] != float64(0) || decoded["diagnostics_truncated"] != false {
t.Fatalf("warning_count = %v", got) t.Fatalf("diagnostic counts = %#v", decoded)
} }
if got := decoded["validation_summaries"]; got != nil { if got := decoded["validation_summaries"]; got != nil {
t.Fatalf("validation_summaries = %#v, want omitted when empty", got) t.Fatalf("validation_summaries = %#v, want omitted when empty", got)
@@ -182,7 +182,14 @@ func testRunOutput() pipeline.RunOutput {
Manifest: artifacts.RunManifest{RunID: "run-123", PipelineID: "sample", ValidationStatus: "rejected"}, Manifest: artifacts.RunManifest{RunID: "run-123", PipelineID: "sample", ValidationStatus: "rejected"},
NormalizeOutputs: []contracts.SerializedOutput{{}, {}}, NormalizeOutputs: []contracts.SerializedOutput{{}, {}},
Rejected: []contracts.RejectedOutput{{}}, Rejected: []contracts.RejectedOutput{{}},
Warnings: []contracts.Warning{{}}, Diagnostics: contracts.DiagnosticCollection{Groups: []contracts.DiagnosticGroup{{
Disposition: contracts.DiagnosticDispositionWarning,
Category: contracts.DiagnosticCategoryFallback,
ReasonCode: "fallback",
Origin: contracts.DiagnosticOrigin{Stage: contracts.DiagnosticOriginStageNormalize, StepID: "step", LaneID: "lane", ModuleKey: "module"},
OccurrenceCount: 1,
Samples: []contracts.DiagnosticSample{{Scope: "scope", Message: "message"}},
}}},
OutputFiles: []contracts.OutputFile{{Name: "index.json"}}, OutputFiles: []contracts.OutputFile{{Name: "index.json"}},
} }
} }

View File

@@ -6,6 +6,7 @@ import (
"io" "io"
"gitea.maximumdirect.net/eric/notarius/internal/core/debugbundle" "gitea.maximumdirect.net/eric/notarius/internal/core/debugbundle"
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline" "gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
) )
@@ -33,13 +34,17 @@ func (s *pipelineCommandState) setDebugPath(debugPath string) {
} }
} }
func (s *pipelineCommandState) observeOutput(output pipeline.RunOutput) { func (s *pipelineCommandState) observeOutput(output pipeline.RunOutput, diagnostics contracts.DiagnosticProjection) {
if s == nil { if s == nil {
return return
} }
s.report.OutputCount = len(output.NormalizeOutputs) s.report.OutputCount = len(output.NormalizeOutputs)
s.report.RejectedCount = len(output.Rejected) s.report.RejectedCount = len(output.Rejected)
s.report.WarningCount = len(output.Warnings) s.report.WarningGroupCount = len(diagnostics.Warnings)
s.report.WarningOccurrenceCount = diagnostics.WarningOccurrenceCount
s.report.DiagnosticGroupCount = len(diagnostics.Diagnostics)
s.report.DiagnosticOccurrenceCount = diagnostics.DiagnosticOccurrenceCount
s.report.DiagnosticsTruncated = output.Diagnostics.Truncated
s.report.ValidationStatus = output.Manifest.ValidationStatus s.report.ValidationStatus = output.Manifest.ValidationStatus
} }

View File

@@ -238,7 +238,7 @@ func TestChangedSemanticSpellCatalogFingerprintCannotResumeRecordedCheckpoint(t
Kind: dnd.SpellListKind, Schema: normalizeSchema, MediaType: "application/json", Content: []byte(`{"spell_casts":[]}`), Kind: dnd.SpellListKind, Schema: normalizeSchema, MediaType: "application/json", Content: []byte(`{"spell_casts":[]}`),
}, },
} }
if err := recorder.NormalizeSucceeded("spells", spellnormalize.Key, normalizeDependencies, normalizeArtifact, nil); err != nil { if err := recorder.NormalizeSucceeded("spells", spellnormalize.Key, normalizeDependencies, normalizeArtifact); err != nil {
t.Fatal(err) t.Fatal(err)
} }
@@ -264,7 +264,7 @@ func TestChangedSemanticSpellCatalogFingerprintCannotResumeRecordedCheckpoint(t
if _, decision := changedLoader.Normalize("spells", spellnormalize.Key, normalizeDependencies); decision.Reused { if _, decision := changedLoader.Normalize("spells", spellnormalize.Key, normalizeDependencies); decision.Reused {
t.Fatalf("changed normalize fingerprint decision = %#v, want normalize checkpoint cold miss", decision) t.Fatalf("changed normalize fingerprint decision = %#v, want normalize checkpoint cold miss", decision)
} }
changedMapping := replaceCheckpointFingerprintValue(t, fingerprints, extractSpellMappingFingerprintName(), "dnd.spells.extract_mapping.v3") changedMapping := replaceCheckpointFingerprintValue(t, fingerprints, extractSpellMappingFingerprintName(), "dnd.spells.extract_mapping.changed")
assertOnlyCheckpointFingerprintChanged(t, fingerprints, changedMapping, extractSpellMappingFingerprintName()) assertOnlyCheckpointFingerprintChanged(t, fingerprints, changedMapping, extractSpellMappingFingerprintName())
_, mappingLoader, err := checkpointHandlersForRun(settings, Options{}, materialized, changedMapping, llmFingerprints, []byte("same input"), nil, nil, "", "", LLMRuntimeOverrides{}, true) _, mappingLoader, err := checkpointHandlersForRun(settings, Options{}, materialized, changedMapping, llmFingerprints, []byte("same input"), nil, nil, "", "", LLMRuntimeOverrides{}, true)
if err != nil { if err != nil {

View File

@@ -24,7 +24,7 @@ func TestProductionSpellCatalogValidationRetries(t *testing.T) {
wantCalls int wantCalls int
wantRejected bool wantRejected bool
wantSpell string wantSpell string
wantWarningCode string wantAdvisoryCode string
}{ }{
{ {
name: "unknown spell remains rejected after exhaustion", name: "unknown spell remains rejected after exhaustion",
@@ -44,7 +44,7 @@ func TestProductionSpellCatalogValidationRetries(t *testing.T) {
}, },
wantCalls: 2, wantCalls: 2,
wantSpell: "Aegis of Emberfall", wantSpell: "Aegis of Emberfall",
wantWarningCode: "spell_not_near_source", wantAdvisoryCode: "spell_not_near_source",
}, },
} }
@@ -92,8 +92,8 @@ func TestProductionSpellCatalogValidationRetries(t *testing.T) {
if rejection.ReasonCode != "unknown_spell" || rejection.AttemptCount != retries+1 { if rejection.ReasonCode != "unknown_spell" || rejection.AttemptCount != retries+1 {
t.Fatalf("rejection = %#v, want exhausted unknown-spell rejection", rejection) t.Fatalf("rejection = %#v, want exhausted unknown-spell rejection", rejection)
} }
if len(output.Warnings) != 1 || output.Warnings[0].ReasonCode != "spell_not_near_source" { if len(output.Diagnostics.Groups) != 1 || output.Diagnostics.Groups[0].ReasonCode != "spell_not_near_source" {
t.Fatalf("warnings = %#v, want complete terminal validation warnings", output.Warnings) t.Fatalf("diagnostics = %#v, want complete terminal validation diagnostic", output.Diagnostics)
} }
return return
} }
@@ -108,8 +108,16 @@ func TestProductionSpellCatalogValidationRetries(t *testing.T) {
if len(value.SpellCasts) != 1 || value.SpellCasts[0].Spell != tt.wantSpell { if len(value.SpellCasts) != 1 || value.SpellCasts[0].Spell != tt.wantSpell {
t.Fatalf("normalized spell list = %#v, want accepted overlay spell", value) t.Fatalf("normalized spell list = %#v, want accepted overlay spell", value)
} }
if len(output.Warnings) != 2 || output.Warnings[0].ReasonCode != tt.wantWarningCode || output.Warnings[1].ReasonCode != tt.wantWarningCode { if len(output.Diagnostics.Groups) != 2 || output.Diagnostics.Groups[0].ReasonCode != tt.wantAdvisoryCode {
t.Fatalf("warnings = %#v, want accepted-attempt warnings from extract and normalize validation", output.Warnings) t.Fatalf("diagnostics = %#v, want terminal extract and normalize diagnostics", output.Diagnostics)
}
if len(output.Diagnostics.Groups) != 2 {
t.Fatalf("diagnostics = %#v, want extract and normalize data-quality advisories", output.Diagnostics)
}
for _, diagnostic := range output.Diagnostics.Groups {
if diagnostic.Disposition != contracts.DiagnosticDispositionAdvisory || diagnostic.ReasonCode != tt.wantAdvisoryCode {
t.Fatalf("diagnostics = %#v, want only data-quality advisories", output.Diagnostics)
}
} }
}) })
} }

View File

@@ -349,7 +349,7 @@ func TestRunUsesOneInjectedIdentityForDebugOutputAndManifest(t *testing.T) {
t.Fatalf("debug invocation session = %q, want %q", invocation.SessionID, wantSessionID) t.Fatalf("debug invocation session = %q, want %q", invocation.SessionID, wantSessionID)
} }
report := readStateTestRunReport(t, debugPath) report := readStateTestRunReport(t, debugPath)
if !report.Succeeded || report.RunID != runID || report.PipelineID != "sample" || report.OutputPath != outputPath || report.DebugPath != debugPath || report.OutputCount != 1 || report.RejectedCount != 0 || report.WarningCount != 0 || report.ValidationStatus != "approved" { if !report.Succeeded || report.RunID != runID || report.PipelineID != "sample" || report.OutputPath != outputPath || report.DebugPath != debugPath || report.OutputCount != 1 || report.RejectedCount != 0 || report.WarningGroupCount != 0 || report.WarningOccurrenceCount != 0 || report.DiagnosticGroupCount != 0 || report.DiagnosticOccurrenceCount != 0 || report.DiagnosticsTruncated || report.ValidationStatus != "approved" {
t.Fatalf("success report = %#v", report) t.Fatalf("success report = %#v", report)
} }
if !strings.Contains(result.stdout, "outputs=1 rejected=0") { if !strings.Contains(result.stdout, "outputs=1 rejected=0") {
@@ -431,7 +431,7 @@ func TestRunWritesTerminalArtifactsForResolutionPipelineAndOutputFailures(t *tes
bundlePath := onlyChildDir(t, roots.debug) bundlePath := onlyChildDir(t, roots.debug)
runID := filepath.Base(bundlePath) runID := filepath.Base(bundlePath)
report := readStateTestRunReport(t, bundlePath) report := readStateTestRunReport(t, bundlePath)
if report.Succeeded || report.RunID != runID || report.PipelineID != tc.pipelineID || report.OutputPath != filepath.Join(roots.output, runID) || report.DebugPath != bundlePath || report.OutputCount != tc.wantOutputs || report.RejectedCount != 0 || report.WarningCount != 0 || report.ValidationStatus != tc.wantValidation { if report.Succeeded || report.RunID != runID || report.PipelineID != tc.pipelineID || report.OutputPath != filepath.Join(roots.output, runID) || report.DebugPath != bundlePath || report.OutputCount != tc.wantOutputs || report.RejectedCount != 0 || report.WarningGroupCount != 0 || report.WarningOccurrenceCount != 0 || report.DiagnosticGroupCount != 0 || report.DiagnosticOccurrenceCount != 0 || report.DiagnosticsTruncated || report.ValidationStatus != tc.wantValidation {
t.Fatalf("failure report = %#v", report) t.Fatalf("failure report = %#v", report)
} }
errorLog, err := os.ReadFile(filepath.Join(bundlePath, "summary", "error.log")) errorLog, err := os.ReadFile(filepath.Join(bundlePath, "summary", "error.log"))
@@ -445,7 +445,7 @@ func TestRunWritesTerminalArtifactsForResolutionPipelineAndOutputFailures(t *tes
func TestRunRetainsPartialPipelineOutcomeInFailureSummary(t *testing.T) { func TestRunRetainsPartialPipelineOutcomeInFailureSummary(t *testing.T) {
roots := newStateTestRoots(t) roots := newStateTestRoots(t)
harness := newStateTestHarness() harness := newStateTestHarness()
harness.chunkWarnings = []contracts.Warning{{Scope: "chunk", ReasonCode: "partial-warning", Message: "warning retained before failure"}} harness.chunkDiagnostics = []contracts.ProducerDiagnostic{stateTestDiagnostic("chunk", "partial-warning", "warning retained before failure")}
harness.extractErr = errors.New("synthetic partial pipeline failure") harness.extractErr = errors.New("synthetic partial pipeline failure")
result := runStateTest(t, roots, harness.options(), true, true, "bypass") result := runStateTest(t, roots, harness.options(), true, true, "bypass")
@@ -454,7 +454,7 @@ func TestRunRetainsPartialPipelineOutcomeInFailureSummary(t *testing.T) {
} }
bundlePath := onlyChildDir(t, roots.debug) bundlePath := onlyChildDir(t, roots.debug)
report := readStateTestRunReport(t, bundlePath) report := readStateTestRunReport(t, bundlePath)
if report.Succeeded || report.OutputCount != 0 || report.RejectedCount != 0 || report.WarningCount != 1 || report.ValidationStatus != "failed" { if report.Succeeded || report.OutputCount != 0 || report.RejectedCount != 0 || report.WarningGroupCount != 1 || report.WarningOccurrenceCount != 1 || report.DiagnosticGroupCount != 0 || report.DiagnosticOccurrenceCount != 0 || report.DiagnosticsTruncated || report.ValidationStatus != "failed" {
t.Fatalf("partial failure report = %#v", report) t.Fatalf("partial failure report = %#v", report)
} }
@@ -463,10 +463,10 @@ func TestRunRetainsPartialPipelineOutcomeInFailureSummary(t *testing.T) {
if manifest.RunID != report.RunID || manifest.PipelineID != "sample" || manifest.ValidationStatus != "failed" { if manifest.RunID != report.RunID || manifest.PipelineID != "sample" || manifest.ValidationStatus != "failed" {
t.Fatalf("partial manifest = %#v", manifest) t.Fatalf("partial manifest = %#v", manifest)
} }
var warnings []contracts.Warning var diagnostics contracts.DiagnosticCollection
readStateTestSummaryJSON(t, bundlePath, "warnings.json", &warnings) readStateTestSummaryJSON(t, bundlePath, "final-diagnostics.json", &diagnostics)
if len(warnings) != 1 || warnings[0].ReasonCode != "partial-warning" { if len(diagnostics.Groups) != 1 || diagnostics.Groups[0].ReasonCode != "partial-warning" {
t.Fatalf("partial warnings = %#v", warnings) t.Fatalf("partial diagnostics = %#v", diagnostics)
} }
var events []pipeline.CheckpointEvent var events []pipeline.CheckpointEvent
readStateTestSummaryJSON(t, bundlePath, "checkpoint-events.json", &events) readStateTestSummaryJSON(t, bundlePath, "checkpoint-events.json", &events)
@@ -863,11 +863,12 @@ type stateTestHarness struct {
chunkCalls, extractCalls int chunkCalls, extractCalls int
runIDCalls uint64 runIDCalls uint64
extractErr error extractErr error
chunkWarnings []contracts.Warning chunkDiagnostics []contracts.ProducerDiagnostic
moduleProfiles []string moduleProfiles []string
sessionIDs []string sessionIDs []string
outputWarnings []contracts.Warning outputDiagnostics contracts.DiagnosticCollection
includeWarnings bool includeWarnings bool
includeWarningFile bool
} }
func newStateTestHarness() *stateTestHarness { return &stateTestHarness{} } func newStateTestHarness() *stateTestHarness { return &stateTestHarness{} }
@@ -892,7 +893,7 @@ func (h *stateTestHarness) options() Options {
panic(err) panic(err)
} }
if err := registries.Outputs.RegisterWithSpec(pipeline.ModuleSpec{Key: "test/output", Stage: pipeline.StageOutput, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"normalized"}, Provides: []string{"output"}}, func() (contracts.OutputEncoder, error) { if err := registries.Outputs.RegisterWithSpec(pipeline.ModuleSpec{Key: "test/output", Stage: pipeline.StageOutput, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"normalized"}, Provides: []string{"output"}}, func() (contracts.OutputEncoder, error) {
return stateTestOutput{harness: h, includeWarnings: h.includeWarnings}, nil return stateTestOutput{harness: h, includeDiagnostics: h.includeWarnings}, nil
}); err != nil { }); err != nil {
panic(err) panic(err)
} }
@@ -931,7 +932,7 @@ func (c stateTestChunker) Plan(_ context.Context, req contracts.ChunkRequest) (c
c.harness.mu.Lock() c.harness.mu.Lock()
c.harness.chunkCalls++ c.harness.chunkCalls++
c.harness.mu.Unlock() c.harness.mu.Unlock()
return contracts.ChunkPlanResult{Plan: source.ChunkPlan{SourceDigest: req.Source.Digest, Ranges: []source.ChunkRange{{StartUnitID: 1, EndUnitID: 1}}}, Warnings: append([]contracts.Warning(nil), c.harness.chunkWarnings...)}, nil return contracts.ChunkPlanResult{Plan: source.ChunkPlan{SourceDigest: req.Source.Digest, Ranges: []source.ChunkRange{{StartUnitID: 1, EndUnitID: 1}}}, Diagnostics: contracts.CloneProducerDiagnostics(c.harness.chunkDiagnostics)}, nil
} }
const stateTestArtifactKind contracts.ArtifactKind = "test/artifact" const stateTestArtifactKind contracts.ArtifactKind = "test/artifact"
@@ -1000,19 +1001,27 @@ func (n stateTestNormalizer) Normalize(_ context.Context, req contracts.TypedNor
type stateTestOutput struct { type stateTestOutput struct {
harness *stateTestHarness harness *stateTestHarness
includeWarnings bool includeDiagnostics bool
} }
func (o stateTestOutput) Key() string { return "test/output" } func (o stateTestOutput) Key() string { return "test/output" }
func (o stateTestOutput) Encode(_ context.Context, req contracts.OutputRequest) (contracts.OutputResult, error) { func (o stateTestOutput) Encode(_ context.Context, req contracts.OutputRequest) (contracts.OutputResult, error) {
o.harness.mu.Lock() o.harness.mu.Lock()
o.harness.outputWarnings = append([]contracts.Warning(nil), req.Warnings...) o.harness.outputDiagnostics = contracts.CloneDiagnosticCollection(req.Diagnostics)
o.harness.mu.Unlock() o.harness.mu.Unlock()
data := []byte("{\"ok\":true}\n") data := []byte("{\"ok\":true}\n")
if o.includeWarnings && len(req.Warnings) > 0 { if o.includeDiagnostics && len(req.Diagnostics.Groups) > 0 {
data = []byte(fmt.Sprintf("{\"ok\":true,\"warnings\":%q}\n", req.Warnings[0].ReasonCode)) data = []byte(fmt.Sprintf("{\"ok\":true,\"diagnostics\":%q}\n", req.Diagnostics.Groups[0].ReasonCode))
} }
return contracts.OutputResult{Files: []contracts.OutputFile{{Name: "result.json", Bytes: data}}}, nil files := []contracts.OutputFile{{Name: "result.json", Bytes: data}}
if o.harness.includeWarningFile {
files = append(files, contracts.OutputFile{Name: "warnings.json", Bytes: []byte("{\"warnings\":true}\n")})
}
return contracts.OutputResult{Files: files}, nil
}
func stateTestDiagnostic(scope, reasonCode, message string) contracts.ProducerDiagnostic {
return contracts.ProducerDiagnostic{Disposition: contracts.DiagnosticDispositionWarning, Category: contracts.DiagnosticCategoryDegradation, ReasonCode: reasonCode, OccurrenceCount: 1, Samples: []contracts.DiagnosticSample{{Scope: scope, Message: message}}}
} }
type failingDebugRecorder struct{} type failingDebugRecorder struct{}

View File

@@ -110,7 +110,7 @@ func TestSummaryWriterWritesEverySummaryArtifact(t *testing.T) {
if err := summary.WriteRunReport(RunReport{RunID: bundle.RunID(), PipelineID: "test"}); err != nil { if err := summary.WriteRunReport(RunReport{RunID: bundle.RunID(), PipelineID: "test"}); err != nil {
t.Fatal(err) t.Fatal(err)
} }
if err := summary.WriteWarnings([]contracts.Warning{{ReasonCode: "test"}}); err != nil { if err := summary.WriteDiagnostics(contracts.DiagnosticCollection{}); err != nil {
t.Fatal(err) t.Fatal(err)
} }
if err := summary.WriteError("failed"); err != nil { if err := summary.WriteError("failed"); err != nil {
@@ -126,7 +126,7 @@ func TestSummaryWriterWritesEverySummaryArtifact(t *testing.T) {
ArtifactRunManifest, ArtifactRunManifest,
ArtifactChunkPlan, ArtifactChunkPlan,
ArtifactRunReport, ArtifactRunReport,
ArtifactWarnings, ArtifactDiagnostics,
ArtifactErrorLog, ArtifactErrorLog,
} { } {
info, err := os.Stat(filepath.Join(bundle.SummaryRoot(), name)) info, err := os.Stat(filepath.Join(bundle.SummaryRoot(), name))

View File

@@ -20,7 +20,7 @@ const (
ArtifactRunManifest = "run-manifest.json" ArtifactRunManifest = "run-manifest.json"
ArtifactChunkPlan = "chunk-plan.json" ArtifactChunkPlan = "chunk-plan.json"
ArtifactRunReport = "run-report.json" ArtifactRunReport = "run-report.json"
ArtifactWarnings = "warnings.json" ArtifactDiagnostics = "final-diagnostics.json"
ArtifactErrorLog = "error.log" ArtifactErrorLog = "error.log"
) )
@@ -52,7 +52,11 @@ type RunReport struct {
Succeeded bool `json:"succeeded"` Succeeded bool `json:"succeeded"`
OutputCount int `json:"output_count"` OutputCount int `json:"output_count"`
RejectedCount int `json:"rejected_count"` RejectedCount int `json:"rejected_count"`
WarningCount int `json:"warning_count"` WarningGroupCount int `json:"warning_group_count"`
WarningOccurrenceCount int `json:"warning_occurrence_count"`
DiagnosticGroupCount int `json:"diagnostic_group_count"`
DiagnosticOccurrenceCount int `json:"diagnostic_occurrence_count"`
DiagnosticsTruncated bool `json:"diagnostics_truncated"`
ValidationStatus string `json:"validation_status,omitempty"` ValidationStatus string `json:"validation_status,omitempty"`
} }
type SummaryWriter struct { type SummaryWriter struct {
@@ -101,8 +105,8 @@ func (w *SummaryWriter) WriteChunkPlan(v artifacts.ChunkPlanSummary) error {
return w.writeJSON(ArtifactChunkPlan, v) return w.writeJSON(ArtifactChunkPlan, v)
} }
func (w *SummaryWriter) WriteRunReport(v RunReport) error { return w.writeJSON(ArtifactRunReport, v) } func (w *SummaryWriter) WriteRunReport(v RunReport) error { return w.writeJSON(ArtifactRunReport, v) }
func (w *SummaryWriter) WriteWarnings(v []contracts.Warning) error { func (w *SummaryWriter) WriteDiagnostics(v contracts.DiagnosticCollection) error {
return w.writeJSON(ArtifactWarnings, v) return w.writeJSON(ArtifactDiagnostics, contracts.CloneDiagnosticCollection(v))
} }
func (w *SummaryWriter) WriteError(message string) error { func (w *SummaryWriter) WriteError(message string) error {
return w.writeBytes(ArtifactErrorLog, []byte(message+"\n")) return w.writeBytes(ArtifactErrorLog, []byte(message+"\n"))

View File

@@ -24,7 +24,6 @@ type filesystemCheckpointFixture struct {
merge pipeline.CheckpointArtifact merge pipeline.CheckpointArtifact
normalize pipeline.CheckpointArtifact normalize pipeline.CheckpointArtifact
dependencies []pipeline.CheckpointFingerprint dependencies []pipeline.CheckpointFingerprint
warnings []contracts.Warning
rejected []contracts.RejectedOutput rejected []contracts.RejectedOutput
} }
@@ -36,9 +35,9 @@ func TestFilesystemCheckpointRoundTripsAllStages(t *testing.T) {
fixture.doc.Metadata["owner"] = "caller mutation" fixture.doc.Metadata["owner"] = "caller mutation"
fixture.extract.Artifact.Content[0] = 'x' fixture.extract.Artifact.Content[0] = 'x'
fixture.extract.Artifact.Metadata["content"] = "caller mutation" fixture.extract.Artifact.Metadata["content"] = "caller mutation"
fixture.extract.Diagnostics[0].Diagnostic.Samples[0].Message = "caller mutation"
fixture.merge.Artifact.Content[0] = 'x' fixture.merge.Artifact.Content[0] = 'x'
fixture.normalize.Artifact.Content[0] = 'x' fixture.normalize.Artifact.Content[0] = 'x'
fixture.warnings[0].Message = "caller mutation"
fixture.rejected[0].Message = "caller mutation" fixture.rejected[0].Message = "caller mutation"
t.Run("source", func(t *testing.T) { t.Run("source", func(t *testing.T) {
@@ -60,11 +59,11 @@ func TestFilesystemCheckpointRoundTripsAllStages(t *testing.T) {
t.Run("extract", func(t *testing.T) { t.Run("extract", func(t *testing.T) {
got, decision := fixture.loader.Extract("lane-a", "extract-module", fixture.dependencies) got, decision := fixture.loader.Extract("lane-a", "extract-module", fixture.dependencies)
if !decision.Reused || len(got.Outputs) != 1 || len(got.Rejected) != 1 || len(got.Warnings) != 1 { if !decision.Reused || len(got.Outputs) != 1 || len(got.Rejected) != 1 || len(got.Outputs[0].Diagnostics) != 1 {
t.Fatalf("extract result=%#v decision=%#v", got, decision) t.Fatalf("extract result=%#v decision=%#v", got, decision)
} }
output := got.Outputs[0] output := got.Outputs[0]
if !bytes.Equal(output.Artifact.Content, []byte(`{"spell":"fire"}`)) || output.Artifact.Kind != "spell" || output.Artifact.Schema.ID != "spell-schema" || output.Artifact.Schema.Version != "1" || output.Artifact.MediaType != "application/json" || output.Artifact.Metadata["chunk"] != "chunk-a" || output.ChunkRef.StartUnitID != 1 || got.Warnings[0].ReasonCode != "partial" || got.Rejected[0].ReasonCode != "invalid_source" { if !bytes.Equal(output.Artifact.Content, []byte(`{"spell":"fire"}`)) || output.Artifact.Kind != "spell" || output.Artifact.Schema.ID != "spell-schema" || output.Artifact.Schema.Version != "1" || output.Artifact.MediaType != "application/json" || output.Artifact.Metadata["chunk"] != "chunk-a" || output.ChunkRef.StartUnitID != 1 || output.Diagnostics[0].Diagnostic.ReasonCode != "normalized_record" || got.Rejected[0].ReasonCode != "invalid_source" {
t.Fatalf("extract values were not restored: %#v", got) t.Fatalf("extract values were not restored: %#v", got)
} }
manifest := readManifest[ExtractLaneManifest](t, filepath.Join(fixture.root, mustRelativePath(t, fixture.identity), "extract", "lane-a", "manifest.json")) manifest := readManifest[ExtractLaneManifest](t, filepath.Join(fixture.root, mustRelativePath(t, fixture.identity), "extract", "lane-a", "manifest.json"))
@@ -73,37 +72,36 @@ func TestFilesystemCheckpointRoundTripsAllStages(t *testing.T) {
} }
got.Outputs[0].Artifact.Content[0] = 'y' got.Outputs[0].Artifact.Content[0] = 'y'
got.Warnings[0].Message = "loaded mutation"
reloaded, decision := fixture.loader.Extract("lane-a", "extract-module", fixture.dependencies) reloaded, decision := fixture.loader.Extract("lane-a", "extract-module", fixture.dependencies)
if !decision.Reused || !bytes.Equal(reloaded.Outputs[0].Artifact.Content, []byte(`{"spell":"fire"}`)) || reloaded.Warnings[0].Message != "partial output" { if !decision.Reused || !bytes.Equal(reloaded.Outputs[0].Artifact.Content, []byte(`{"spell":"fire"}`)) {
t.Fatalf("extract reload changed after loaded mutation: %#v decision=%#v", reloaded, decision) t.Fatalf("extract reload changed after loaded mutation: %#v decision=%#v", reloaded, decision)
} }
}) })
for _, tt := range []struct { for _, tt := range []struct {
name string name string
load func() (pipeline.CheckpointArtifact, []contracts.Warning, pipeline.CheckpointDecision) load func() (pipeline.CheckpointArtifact, pipeline.CheckpointDecision)
want []byte want []byte
}{ }{
{name: "merge", load: func() (pipeline.CheckpointArtifact, []contracts.Warning, pipeline.CheckpointDecision) { {name: "merge", load: func() (pipeline.CheckpointArtifact, pipeline.CheckpointDecision) {
got, decision := fixture.loader.Merge("lane-a", "merge-module", fixture.dependencies) got, decision := fixture.loader.Merge("lane-a", "merge-module", fixture.dependencies)
return got.Output, got.Warnings, decision return got.Output, decision
}, want: []byte(`{"spells":["fire"]}`)}, }, want: []byte(`{"spells":["fire"]}`)},
{name: "normalize", load: func() (pipeline.CheckpointArtifact, []contracts.Warning, pipeline.CheckpointDecision) { {name: "normalize", load: func() (pipeline.CheckpointArtifact, pipeline.CheckpointDecision) {
got, decision := fixture.loader.Normalize("lane-a", "normalize-module", fixture.dependencies) got, decision := fixture.loader.Normalize("lane-a", "normalize-module", fixture.dependencies)
return got.Output, got.Warnings, decision return got.Output, decision
}, want: []byte(`{"spells":["fire"],"normalized":true}`)}, }, want: []byte(`{"spells":["fire"],"normalized":true}`)},
} { } {
t.Run(tt.name, func(t *testing.T) { t.Run(tt.name, func(t *testing.T) {
got, warnings, decision := tt.load() got, decision := tt.load()
if !decision.Reused || !bytes.Equal(got.Artifact.Content, tt.want) || got.Artifact.Kind != "spell" || got.Artifact.Schema.ID != "spell-schema" || got.Artifact.Schema.Version != "1" || got.Artifact.Metadata["lane"] != "lane-a" || len(warnings) != 1 || warnings[0].ReasonCode != "review" { if !decision.Reused || !bytes.Equal(got.Artifact.Content, tt.want) || got.Artifact.Kind != "spell" || got.Artifact.Schema.ID != "spell-schema" || got.Artifact.Schema.Version != "1" || got.Artifact.Metadata["lane"] != "lane-a" || len(got.Diagnostics) != 1 || got.Diagnostics[0].Diagnostic.ReasonCode != "normalized_record" {
t.Fatalf("%s result=%#v warnings=%#v decision=%#v", tt.name, got, warnings, decision) t.Fatalf("%s result=%#v decision=%#v", tt.name, got, decision)
} }
got.Artifact.Content[0] = 'z' got.Artifact.Content[0] = 'z'
reloaded, warnings, decision := tt.load() reloaded, decision := tt.load()
if !decision.Reused || !bytes.Equal(reloaded.Artifact.Content, tt.want) || warnings[0].Message != "review manually" { if !decision.Reused || !bytes.Equal(reloaded.Artifact.Content, tt.want) {
t.Fatalf("%s reload changed after loaded mutation: %#v warnings=%#v decision=%#v", tt.name, reloaded, warnings, decision) t.Fatalf("%s reload changed after loaded mutation: %#v decision=%#v", tt.name, reloaded, decision)
} }
}) })
} }
@@ -159,6 +157,7 @@ func TestFilesystemCheckpointRejectsIncompatibleManifests(t *testing.T) {
}{ }{
{"v1 schema", func(m map[string]any) { m["workspace_schema_version"] = WorkspaceSchemaVersionV1 }, pipeline.CheckpointReasonWorkspaceSchemaIncompatible}, {"v1 schema", func(m map[string]any) { m["workspace_schema_version"] = WorkspaceSchemaVersionV1 }, pipeline.CheckpointReasonWorkspaceSchemaIncompatible},
{"v2 schema", func(m map[string]any) { m["workspace_schema_version"] = WorkspaceSchemaVersionV2 }, pipeline.CheckpointReasonWorkspaceSchemaIncompatible}, {"v2 schema", func(m map[string]any) { m["workspace_schema_version"] = WorkspaceSchemaVersionV2 }, pipeline.CheckpointReasonWorkspaceSchemaIncompatible},
{"v3 schema", func(m map[string]any) { m["workspace_schema_version"] = WorkspaceSchemaVersionV3 }, pipeline.CheckpointReasonWorkspaceSchemaIncompatible},
{"unknown schema", func(m map[string]any) { m["workspace_schema_version"] = "notarius.workspace.future" }, pipeline.CheckpointReasonWorkspaceSchemaIncompatible}, {"unknown schema", func(m map[string]any) { m["workspace_schema_version"] = "notarius.workspace.future" }, pipeline.CheckpointReasonWorkspaceSchemaIncompatible},
{"identity", func(m map[string]any) { m["metadata"].(map[string]any)["checkpoint_identity_digest"] = "sha256:other" }, pipeline.CheckpointReasonIdentityMismatch}, {"identity", func(m map[string]any) { m["metadata"].(map[string]any)["checkpoint_identity_digest"] = "sha256:other" }, pipeline.CheckpointReasonIdentityMismatch},
{"stage", func(m map[string]any) { m["stage"] = string(StageMerge) }, pipeline.CheckpointReasonStageMismatch}, {"stage", func(m map[string]any) { m["stage"] = string(StageMerge) }, pipeline.CheckpointReasonStageMismatch},
@@ -204,6 +203,11 @@ func TestFilesystemCheckpointRejectsIncompleteArtifactsAndContent(t *testing.T)
{"content digest", func(m map[string]any) { {"content digest", func(m map[string]any) {
m["outputs"].([]any)[0].(map[string]any)["content"].(map[string]any)["content_digest"] = "sha256:other" m["outputs"].([]any)[0].(map[string]any)["content"].(map[string]any)["content_digest"] = "sha256:other"
}, pipeline.CheckpointReasonArtifactDigestMismatch}, }, pipeline.CheckpointReasonArtifactDigestMismatch},
{"diagnostics", func(m map[string]any) {
m["outputs"].([]any)[0].(map[string]any)["diagnostics"] = []any{map[string]any{
"diagnostic": map[string]any{"disposition": "warning", "category": "configuration", "reason_code": "invalid", "occurrence_count": float64(0)},
}}
}, pipeline.CheckpointReasonArtifactPayloadInvalid},
} { } {
t.Run(tt.name, func(t *testing.T) { t.Run(tt.name, func(t *testing.T) {
fixture := seedFilesystemCheckpoints(t) fixture := seedFilesystemCheckpoints(t)
@@ -337,9 +341,6 @@ func TestFilesystemLoaderReadsAcceptedNormalizeWithoutStageDependencies(t *testi
if checkpoint.Output.Artifact.Content == nil || string(checkpoint.Output.Artifact.Content) != string(fixture.normalize.Artifact.Content) { if checkpoint.Output.Artifact.Content == nil || string(checkpoint.Output.Artifact.Content) != string(fixture.normalize.Artifact.Content) {
t.Fatalf("accepted normalize output = %#v, want recorded artifact", checkpoint.Output) t.Fatalf("accepted normalize output = %#v, want recorded artifact", checkpoint.Output)
} }
if len(checkpoint.Warnings) != 1 || checkpoint.Warnings[0].ReasonCode != "normalized" {
t.Fatalf("accepted normalize warnings = %#v", checkpoint.Warnings)
}
} }
func TestFilesystemLoaderRejectsInvalidAcceptedNormalize(t *testing.T) { func TestFilesystemLoaderRejectsInvalidAcceptedNormalize(t *testing.T) {
@@ -402,8 +403,7 @@ func seedAcceptedNormalizeCheckpoint(t *testing.T) filesystemCheckpointFixture {
t.Fatal(err) t.Fatal(err)
} }
stepRecorder := recorder.(pipeline.StepCheckpointRecorder) stepRecorder := recorder.(pipeline.StepCheckpointRecorder)
warnings := []contracts.Warning{{Scope: "normalize", ReasonCode: "normalized", Message: "normalized warning"}} if err := stepRecorder.NormalizeSucceededForStep("step-1", "lane-a", "normalize-module", []pipeline.CheckpointFingerprint{{Name: "merge", Value: "sha256:unavailable"}}, artifact); err != nil {
if err := stepRecorder.NormalizeSucceededForStep("step-1", "lane-a", "normalize-module", []pipeline.CheckpointFingerprint{{Name: "merge", Value: "sha256:unavailable"}}, artifact, warnings); err != nil {
t.Fatal(err) t.Fatal(err)
} }
loader, err := NewFilesystemLoader(root, identity) loader, err := NewFilesystemLoader(root, identity)
@@ -493,20 +493,18 @@ func seedFilesystemCheckpoints(t *testing.T) filesystemCheckpointFixture {
merge: checkpointArtifact("merge", `{"spells":["fire"]}`), merge: checkpointArtifact("merge", `{"spells":["fire"]}`),
normalize: checkpointArtifact("normalize", `{"spells":["fire"],"normalized":true}`), normalize: checkpointArtifact("normalize", `{"spells":["fire"],"normalized":true}`),
dependencies: []pipeline.CheckpointFingerprint{{Name: "source", Value: "sha256:source"}, {Name: "chunk-plan", Value: "sha256:plan"}}, dependencies: []pipeline.CheckpointFingerprint{{Name: "source", Value: "sha256:source"}, {Name: "chunk-plan", Value: "sha256:plan"}},
warnings: []contracts.Warning{{Scope: "extract", ReasonCode: "partial", Message: "partial output"}},
rejected: []contracts.RejectedOutput{{Stage: "extract", LaneID: "lane-a", ModuleKey: "extract-module", ChunkID: "chunk-a", ValidatorName: "source_refs", ReasonCode: "invalid_source", Message: "source reference is invalid", AttemptCount: 1}}, rejected: []contracts.RejectedOutput{{Stage: "extract", LaneID: "lane-a", ModuleKey: "extract-module", ChunkID: "chunk-a", ValidatorName: "source_refs", ReasonCode: "invalid_source", Message: "source reference is invalid", AttemptCount: 1}},
} }
if err := recorder.SourceSucceeded("source-module", &fixture.doc); err != nil { if err := recorder.SourceSucceeded("source-module", &fixture.doc); err != nil {
t.Fatal(err) t.Fatal(err)
} }
if err := recorder.ExtractSucceeded("lane-a", "extract-module", fixture.dependencies, []pipeline.CheckpointArtifact{fixture.extract}, fixture.rejected, fixture.warnings); err != nil { if err := recorder.ExtractSucceeded("lane-a", "extract-module", fixture.dependencies, []pipeline.CheckpointArtifact{fixture.extract}, fixture.rejected); err != nil {
t.Fatal(err) t.Fatal(err)
} }
mergeWarnings := []contracts.Warning{{Scope: "merge", ReasonCode: "review", Message: "review manually"}} if err := recorder.MergeSucceeded("lane-a", "merge-module", fixture.dependencies, fixture.merge); err != nil {
if err := recorder.MergeSucceeded("lane-a", "merge-module", fixture.dependencies, fixture.merge, mergeWarnings); err != nil {
t.Fatal(err) t.Fatal(err)
} }
if err := recorder.NormalizeSucceeded("lane-a", "normalize-module", fixture.dependencies, fixture.normalize, mergeWarnings); err != nil { if err := recorder.NormalizeSucceeded("lane-a", "normalize-module", fixture.dependencies, fixture.normalize); err != nil {
t.Fatal(err) t.Fatal(err)
} }
fixture.loader, err = NewFilesystemLoader(root, identity) fixture.loader, err = NewFilesystemLoader(root, identity)
@@ -528,6 +526,7 @@ func checkpointArtifact(module, content string) pipeline.CheckpointArtifact {
LaneID: "lane-a", ModuleKey: module, SourceID: "document-1", ChunkID: "chunk-a", ChunkIndex: 0, LaneID: "lane-a", ModuleKey: module, SourceID: "document-1", ChunkID: "chunk-a", ChunkIndex: 0,
ChunkRef: source.SourceRef{SourceID: "document-1", StartUnitID: 1, EndUnitID: 1}, SchemaDigest: "sha256:schema", ChunkRef: source.SourceRef{SourceID: "document-1", StartUnitID: 1, EndUnitID: 1}, SchemaDigest: "sha256:schema",
Artifact: contracts.SerializedArtifact{Kind: "spell", Schema: contracts.ArtifactSchema{ID: "spell-schema", Name: "Spell", Version: "1", JSONSchema: []byte(`{"type":"object"}`)}, MediaType: "application/json", Content: []byte(content), Metadata: map[string]any{"chunk": "chunk-a", "lane": "lane-a"}}, Artifact: contracts.SerializedArtifact{Kind: "spell", Schema: contracts.ArtifactSchema{ID: "spell-schema", Name: "Spell", Version: "1", JSONSchema: []byte(`{"type":"object"}`)}, MediaType: "application/json", Content: []byte(content), Metadata: map[string]any{"chunk": "chunk-a", "lane": "lane-a"}},
Diagnostics: []pipeline.CheckpointDiagnostic{{Diagnostic: contracts.ProducerDiagnostic{Disposition: contracts.DiagnosticDispositionObservation, Category: contracts.DiagnosticCategoryNormalization, ReasonCode: "normalized_record", OccurrenceCount: 1, Samples: []contracts.DiagnosticSample{{Scope: "fixture", Message: "record normalized"}}}}},
} }
} }

View File

@@ -86,7 +86,7 @@ func (l *FilesystemLoader) ExtractForStep(stepID, laneID, moduleKey string, depe
if !fingerprintsEqual(checkpointToPipelineFingerprints(manifest.OutputDigests), artifactOutputDigests(outputs)) { if !fingerprintsEqual(checkpointToPipelineFingerprints(manifest.OutputDigests), artifactOutputDigests(outputs)) {
return pipeline.ExtractCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactDigestMismatch) return pipeline.ExtractCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactDigestMismatch)
} }
return pipeline.ExtractCheckpoint{Outputs: outputs, Rejected: cloneRejectedOutputs(payload.Rejected), Warnings: cloneWarnings(payload.Warnings)}, reusedDecision() return pipeline.ExtractCheckpoint{Outputs: outputs, Rejected: cloneRejectedOutputs(payload.Rejected)}, reusedDecision()
} }
func (l *FilesystemLoader) Merge(laneID, moduleKey string, dependencies []pipeline.CheckpointFingerprint) (pipeline.MergeCheckpoint, pipeline.CheckpointDecision) { func (l *FilesystemLoader) Merge(laneID, moduleKey string, dependencies []pipeline.CheckpointFingerprint) (pipeline.MergeCheckpoint, pipeline.CheckpointDecision) {
@@ -112,7 +112,7 @@ func (l *FilesystemLoader) MergeForStep(stepID, laneID, moduleKey string, depend
if !fingerprintsEqual(checkpointToPipelineFingerprints(manifest.OutputDigests), artifactOutputDigests(values)) { if !fingerprintsEqual(checkpointToPipelineFingerprints(manifest.OutputDigests), artifactOutputDigests(values)) {
return pipeline.MergeCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactDigestMismatch) return pipeline.MergeCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactDigestMismatch)
} }
return pipeline.MergeCheckpoint{Output: values[0], Warnings: cloneWarnings(payload.Warnings)}, reusedDecision() return pipeline.MergeCheckpoint{Output: values[0]}, reusedDecision()
} }
func (l *FilesystemLoader) Normalize(laneID, moduleKey string, dependencies []pipeline.CheckpointFingerprint) (pipeline.NormalizeCheckpoint, pipeline.CheckpointDecision) { func (l *FilesystemLoader) Normalize(laneID, moduleKey string, dependencies []pipeline.CheckpointFingerprint) (pipeline.NormalizeCheckpoint, pipeline.CheckpointDecision) {
@@ -138,7 +138,7 @@ func (l *FilesystemLoader) NormalizeForStep(stepID, laneID, moduleKey string, de
if !fingerprintsEqual(checkpointToPipelineFingerprints(manifest.OutputDigests), artifactOutputDigests(values)) { if !fingerprintsEqual(checkpointToPipelineFingerprints(manifest.OutputDigests), artifactOutputDigests(values)) {
return pipeline.NormalizeCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactDigestMismatch) return pipeline.NormalizeCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactDigestMismatch)
} }
return pipeline.NormalizeCheckpoint{Output: values[0], Warnings: cloneWarnings(payload.Warnings)}, reusedDecision() return pipeline.NormalizeCheckpoint{Output: values[0]}, reusedDecision()
} }
func (l *FilesystemLoader) AcceptedNormalize(stepID, laneID, moduleKey string) (pipeline.NormalizeCheckpoint, pipeline.CheckpointDecision) { func (l *FilesystemLoader) AcceptedNormalize(stepID, laneID, moduleKey string) (pipeline.NormalizeCheckpoint, pipeline.CheckpointDecision) {
@@ -163,7 +163,7 @@ func (l *FilesystemLoader) AcceptedNormalize(stepID, laneID, moduleKey string) (
if !fingerprintsEqual(checkpointToPipelineFingerprints(manifest.OutputDigests), artifactOutputDigests(values)) { if !fingerprintsEqual(checkpointToPipelineFingerprints(manifest.OutputDigests), artifactOutputDigests(values)) {
return pipeline.NormalizeCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactDigestMismatch) return pipeline.NormalizeCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactDigestMismatch)
} }
return pipeline.NormalizeCheckpoint{Output: values[0], Warnings: cloneWarnings(payload.Warnings)}, decision(pipeline.CheckpointDecisionReused, pipeline.CheckpointReasonAcceptedArtifactReused) return pipeline.NormalizeCheckpoint{Output: values[0]}, decision(pipeline.CheckpointDecisionReused, pipeline.CheckpointReasonAcceptedArtifactReused)
} }
func (l *FilesystemLoader) validateAcceptedNormalizeManifest(manifest StageManifest, stepID, laneID, moduleKey string) pipeline.CheckpointDecision { func (l *FilesystemLoader) validateAcceptedNormalizeManifest(manifest StageManifest, stepID, laneID, moduleKey string) pipeline.CheckpointDecision {
@@ -205,11 +205,33 @@ func artifactCheckpointOutputs(values []artifactCheckpointEnvelope) ([]pipeline.
if strings.TrimSpace(string(v.Kind)) == "" || strings.TrimSpace(v.Schema.ID) == "" || strings.TrimSpace(v.Schema.Version) == "" || strings.TrimSpace(v.SchemaDigest) == "" { if strings.TrimSpace(string(v.Kind)) == "" || strings.TrimSpace(v.Schema.ID) == "" || strings.TrimSpace(v.Schema.Version) == "" || strings.TrimSpace(v.SchemaDigest) == "" {
return nil, &artifactPayloadError{code: pipeline.CheckpointReasonArtifactCodecIncompatible, err: fmt.Errorf("artifact codec identity is incomplete")} return nil, &artifactPayloadError{code: pipeline.CheckpointReasonArtifactCodecIncompatible, err: fmt.Errorf("artifact codec identity is incomplete")}
} }
out = append(out, pipeline.CheckpointArtifact{LaneID: v.LaneID, ModuleKey: v.ModuleKey, SourceID: v.SourceID, ChunkID: v.ChunkID, ChunkIndex: v.ChunkIndex, ChunkRef: v.ChunkRef, SchemaDigest: v.SchemaDigest, Artifact: contracts.SerializedArtifact{Kind: v.Kind, Schema: v.Schema, MediaType: v.Content.MediaType, Content: content, Metadata: cloneMetadata(v.Content.Metadata)}}) diagnostics, err := cloneAndValidateCheckpointDiagnostics(v.Diagnostics)
if err != nil {
return nil, err
}
out = append(out, pipeline.CheckpointArtifact{LaneID: v.LaneID, ModuleKey: v.ModuleKey, SourceID: v.SourceID, ChunkID: v.ChunkID, ChunkIndex: v.ChunkIndex, ChunkRef: v.ChunkRef, SchemaDigest: v.SchemaDigest, Artifact: contracts.SerializedArtifact{Kind: v.Kind, Schema: v.Schema, MediaType: v.Content.MediaType, Content: content, Metadata: cloneMetadata(v.Content.Metadata)}, Diagnostics: diagnostics})
} }
return out, nil return out, nil
} }
func cloneAndValidateCheckpointDiagnostics(values []pipeline.CheckpointDiagnostic) ([]pipeline.CheckpointDiagnostic, error) {
if len(values) == 0 {
return nil, nil
}
diagnostics := make([]contracts.ProducerDiagnostic, len(values))
for index, value := range values {
diagnostics[index] = value.Diagnostic
}
if err := contracts.ValidateProducerDiagnostics(diagnostics); err != nil {
return nil, &artifactPayloadError{code: pipeline.CheckpointReasonArtifactPayloadInvalid, err: fmt.Errorf("checkpoint diagnostics: %w", err)}
}
cloned := make([]pipeline.CheckpointDiagnostic, len(values))
for index, value := range values {
cloned[index] = pipeline.CheckpointDiagnostic{Diagnostic: contracts.CloneProducerDiagnostics([]contracts.ProducerDiagnostic{value.Diagnostic})[0], ValidatorKey: value.ValidatorKey}
}
return cloned, nil
}
func (l *FilesystemLoader) readJSON(name string, out any) pipeline.CheckpointDecision { func (l *FilesystemLoader) readJSON(name string, out any) pipeline.CheckpointDecision {
if !l.Enabled() { if !l.Enabled() {
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonLoadingDisabled) return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonLoadingDisabled)

View File

@@ -3,7 +3,8 @@ package checkpoint
import "time" import "time"
const ( const (
WorkspaceSchemaVersion = "notarius.workspace.v3" WorkspaceSchemaVersion = "notarius.workspace.v4"
WorkspaceSchemaVersionV3 = "notarius.workspace.v3"
WorkspaceSchemaVersionV2 = "notarius.workspace.v2" WorkspaceSchemaVersionV2 = "notarius.workspace.v2"
WorkspaceSchemaVersionV1 = "notarius.workspace.v1" WorkspaceSchemaVersionV1 = "notarius.workspace.v1"
) )

View File

@@ -81,18 +81,18 @@ func (r *FilesystemRecorder) ExtractRunningForStep(stepID, laneID string, module
return r.writeManifest(laneManifestPath("extract", stepID, laneID), ExtractLaneManifest{StageManifest: manifest}) return r.writeManifest(laneManifestPath("extract", stepID, laneID), ExtractLaneManifest{StageManifest: manifest})
} }
func (r *FilesystemRecorder) ExtractSucceeded(laneID, moduleKey string, dependencies []pipeline.CheckpointFingerprint, outputs []pipeline.CheckpointArtifact, rejected []contracts.RejectedOutput, warnings []contracts.Warning) error { func (r *FilesystemRecorder) ExtractSucceeded(laneID, moduleKey string, dependencies []pipeline.CheckpointFingerprint, outputs []pipeline.CheckpointArtifact, rejected []contracts.RejectedOutput) error {
return r.ExtractSucceededForStep("", laneID, moduleKey, dependencies, outputs, rejected, warnings) return r.ExtractSucceededForStep("", laneID, moduleKey, dependencies, outputs, rejected)
} }
func (r *FilesystemRecorder) ExtractSucceededForStep(stepID, laneID, moduleKey string, dependencies []pipeline.CheckpointFingerprint, outputs []pipeline.CheckpointArtifact, rejected []contracts.RejectedOutput, warnings []contracts.Warning) error { func (r *FilesystemRecorder) ExtractSucceededForStep(stepID, laneID, moduleKey string, dependencies []pipeline.CheckpointFingerprint, outputs []pipeline.CheckpointArtifact, rejected []contracts.RejectedOutput) error {
payload := artifactExtractEnvelope{Outputs: artifactCheckpointEnvelopes(outputs), Rejected: cloneRejectedOutputs(rejected), Warnings: cloneWarnings(warnings)} payload := artifactExtractEnvelope{Outputs: artifactCheckpointEnvelopes(outputs), Rejected: cloneRejectedOutputs(rejected)}
if err := r.writePayload(lanePayloadPath("extract", stepID, laneID, "outputs.json"), payload); err != nil { if err := r.writePayload(lanePayloadPath("extract", stepID, laneID, "outputs.json"), payload); err != nil {
return err return err
} }
manifest := r.laneManifest(StageExtract, statusForRejected(rejected), stepID, laneID, moduleKey, dependencies) manifest := r.laneManifest(StageExtract, statusForRejected(rejected), stepID, laneID, moduleKey, dependencies)
manifest.OutputDigests = checkpointFingerprints(artifactOutputDigests(outputs)) manifest.OutputDigests = checkpointFingerprints(artifactOutputDigests(outputs))
manifest.ValidationStatus = validationStatusString(warnings, rejected) manifest.ValidationStatus = validationStatusString(rejected)
manifest.Rejections = rejectionSummaries(rejected) manifest.Rejections = rejectionSummaries(rejected)
manifest.CompletedAt = timePtr(r.timestamp()) manifest.CompletedAt = timePtr(r.timestamp())
return r.writeManifest(laneManifestPath("extract", stepID, laneID), ExtractLaneManifest{StageManifest: manifest, ChunkCount: len(outputs) + len(rejected), OutputCount: len(outputs)}) return r.writeManifest(laneManifestPath("extract", stepID, laneID), ExtractLaneManifest{StageManifest: manifest, ChunkCount: len(outputs) + len(rejected), OutputCount: len(outputs)})
@@ -119,17 +119,17 @@ func (r *FilesystemRecorder) MergeRunningForStep(stepID, laneID string, moduleKe
return r.writeManifest(laneManifestPath("merge", stepID, laneID), MergeLaneManifest{StageManifest: manifest}) return r.writeManifest(laneManifestPath("merge", stepID, laneID), MergeLaneManifest{StageManifest: manifest})
} }
func (r *FilesystemRecorder) MergeSucceeded(laneID, moduleKey string, dependencies []pipeline.CheckpointFingerprint, output pipeline.CheckpointArtifact, warnings []contracts.Warning) error { func (r *FilesystemRecorder) MergeSucceeded(laneID, moduleKey string, dependencies []pipeline.CheckpointFingerprint, output pipeline.CheckpointArtifact) error {
return r.MergeSucceededForStep("", laneID, moduleKey, dependencies, output, warnings) return r.MergeSucceededForStep("", laneID, moduleKey, dependencies, output)
} }
func (r *FilesystemRecorder) MergeSucceededForStep(stepID, laneID, moduleKey string, dependencies []pipeline.CheckpointFingerprint, output pipeline.CheckpointArtifact, warnings []contracts.Warning) error { func (r *FilesystemRecorder) MergeSucceededForStep(stepID, laneID, moduleKey string, dependencies []pipeline.CheckpointFingerprint, output pipeline.CheckpointArtifact) error {
if err := r.writePayload(lanePayloadPath("merge", stepID, laneID, "output.json"), artifactSingleEnvelope{Output: artifactCheckpointEnvelopeFromOutput(output), Warnings: cloneWarnings(warnings)}); err != nil { if err := r.writePayload(lanePayloadPath("merge", stepID, laneID, "output.json"), artifactSingleEnvelope{Output: artifactCheckpointEnvelopeFromOutput(output)}); err != nil {
return err return err
} }
manifest := r.laneManifest(StageMerge, StatusSucceeded, stepID, laneID, moduleKey, dependencies) manifest := r.laneManifest(StageMerge, StatusSucceeded, stepID, laneID, moduleKey, dependencies)
manifest.OutputDigests = checkpointFingerprints(artifactOutputDigests([]pipeline.CheckpointArtifact{output})) manifest.OutputDigests = checkpointFingerprints(artifactOutputDigests([]pipeline.CheckpointArtifact{output}))
manifest.ValidationStatus = validationStatusString(warnings, nil) manifest.ValidationStatus = validationStatusString(nil)
manifest.CompletedAt = timePtr(r.timestamp()) manifest.CompletedAt = timePtr(r.timestamp())
return r.writeManifest(laneManifestPath("merge", stepID, laneID), MergeLaneManifest{StageManifest: manifest, InputCount: len(dependencies)}) return r.writeManifest(laneManifestPath("merge", stepID, laneID), MergeLaneManifest{StageManifest: manifest, InputCount: len(dependencies)})
} }
@@ -167,17 +167,17 @@ func (r *FilesystemRecorder) NormalizeRunningForStep(stepID, laneID string, modu
return r.writeManifest(laneManifestPath("normalize", stepID, laneID), NormalizeLaneManifest{StageManifest: manifest}) return r.writeManifest(laneManifestPath("normalize", stepID, laneID), NormalizeLaneManifest{StageManifest: manifest})
} }
func (r *FilesystemRecorder) NormalizeSucceeded(laneID, moduleKey string, dependencies []pipeline.CheckpointFingerprint, output pipeline.CheckpointArtifact, warnings []contracts.Warning) error { func (r *FilesystemRecorder) NormalizeSucceeded(laneID, moduleKey string, dependencies []pipeline.CheckpointFingerprint, output pipeline.CheckpointArtifact) error {
return r.NormalizeSucceededForStep("", laneID, moduleKey, dependencies, output, warnings) return r.NormalizeSucceededForStep("", laneID, moduleKey, dependencies, output)
} }
func (r *FilesystemRecorder) NormalizeSucceededForStep(stepID, laneID, moduleKey string, dependencies []pipeline.CheckpointFingerprint, output pipeline.CheckpointArtifact, warnings []contracts.Warning) error { func (r *FilesystemRecorder) NormalizeSucceededForStep(stepID, laneID, moduleKey string, dependencies []pipeline.CheckpointFingerprint, output pipeline.CheckpointArtifact) error {
if err := r.writePayload(lanePayloadPath("normalize", stepID, laneID, "output.json"), artifactSingleEnvelope{Output: artifactCheckpointEnvelopeFromOutput(output), Warnings: cloneWarnings(warnings)}); err != nil { if err := r.writePayload(lanePayloadPath("normalize", stepID, laneID, "output.json"), artifactSingleEnvelope{Output: artifactCheckpointEnvelopeFromOutput(output)}); err != nil {
return err return err
} }
manifest := r.laneManifest(StageNormalize, StatusSucceeded, stepID, laneID, moduleKey, dependencies) manifest := r.laneManifest(StageNormalize, StatusSucceeded, stepID, laneID, moduleKey, dependencies)
manifest.OutputDigests = checkpointFingerprints(artifactOutputDigests([]pipeline.CheckpointArtifact{output})) manifest.OutputDigests = checkpointFingerprints(artifactOutputDigests([]pipeline.CheckpointArtifact{output}))
manifest.ValidationStatus = validationStatusString(warnings, nil) manifest.ValidationStatus = validationStatusString(nil)
manifest.CompletedAt = timePtr(r.timestamp()) manifest.CompletedAt = timePtr(r.timestamp())
return r.writeManifest(laneManifestPath("normalize", stepID, laneID), NormalizeLaneManifest{StageManifest: manifest, InputCount: len(dependencies)}) return r.writeManifest(laneManifestPath("normalize", stepID, laneID), NormalizeLaneManifest{StageManifest: manifest, InputCount: len(dependencies)})
} }
@@ -253,7 +253,6 @@ type binaryEnvelope struct {
ContentDigest string `json:"content_digest,omitempty"` ContentDigest string `json:"content_digest,omitempty"`
MediaType string `json:"media_type,omitempty"` MediaType string `json:"media_type,omitempty"`
Metadata map[string]any `json:"metadata,omitempty"` Metadata map[string]any `json:"metadata,omitempty"`
Warnings []contracts.Warning `json:"warnings,omitempty"`
} }
type artifactCheckpointEnvelope struct { type artifactCheckpointEnvelope struct {
@@ -267,21 +266,20 @@ type artifactCheckpointEnvelope struct {
Schema contracts.ArtifactSchema `json:"schema"` Schema contracts.ArtifactSchema `json:"schema"`
SchemaDigest string `json:"schema_digest"` SchemaDigest string `json:"schema_digest"`
Content binaryEnvelope `json:"content"` Content binaryEnvelope `json:"content"`
Diagnostics []pipeline.CheckpointDiagnostic `json:"diagnostics,omitempty"`
} }
type artifactExtractEnvelope struct { type artifactExtractEnvelope struct {
Outputs []artifactCheckpointEnvelope `json:"outputs"` Outputs []artifactCheckpointEnvelope `json:"outputs"`
Rejected []contracts.RejectedOutput `json:"rejected,omitempty"` Rejected []contracts.RejectedOutput `json:"rejected,omitempty"`
Warnings []contracts.Warning `json:"warnings,omitempty"`
} }
type artifactSingleEnvelope struct { type artifactSingleEnvelope struct {
Output artifactCheckpointEnvelope `json:"output"` Output artifactCheckpointEnvelope `json:"output"`
Warnings []contracts.Warning `json:"warnings,omitempty"`
} }
func artifactCheckpointEnvelopeFromOutput(output pipeline.CheckpointArtifact) artifactCheckpointEnvelope { func artifactCheckpointEnvelopeFromOutput(output pipeline.CheckpointArtifact) artifactCheckpointEnvelope {
schema := contracts.CloneArtifactSchema(output.Artifact.Schema) schema := contracts.CloneArtifactSchema(output.Artifact.Schema)
schema.JSONSchema = nil schema.JSONSchema = nil
return artifactCheckpointEnvelope{LaneID: output.LaneID, ModuleKey: output.ModuleKey, SourceID: output.SourceID, ChunkID: output.ChunkID, ChunkIndex: output.ChunkIndex, ChunkRef: output.ChunkRef, Kind: output.Artifact.Kind, Schema: schema, SchemaDigest: output.SchemaDigest, Content: binaryEnvelopeFromContent(output.Artifact.Content, output.Artifact.MediaType, output.Artifact.Metadata, nil)} return artifactCheckpointEnvelope{LaneID: output.LaneID, ModuleKey: output.ModuleKey, SourceID: output.SourceID, ChunkID: output.ChunkID, ChunkIndex: output.ChunkIndex, ChunkRef: output.ChunkRef, Kind: output.Artifact.Kind, Schema: schema, SchemaDigest: output.SchemaDigest, Content: binaryEnvelopeFromContent(output.Artifact.Content, output.Artifact.MediaType, output.Artifact.Metadata), Diagnostics: cloneCheckpointDiagnostics(output.Diagnostics)}
} }
func artifactCheckpointEnvelopes(outputs []pipeline.CheckpointArtifact) []artifactCheckpointEnvelope { func artifactCheckpointEnvelopes(outputs []pipeline.CheckpointArtifact) []artifactCheckpointEnvelope {
if len(outputs) == 0 { if len(outputs) == 0 {
@@ -293,6 +291,20 @@ func artifactCheckpointEnvelopes(outputs []pipeline.CheckpointArtifact) []artifa
} }
return out return out
} }
func cloneCheckpointDiagnostics(values []pipeline.CheckpointDiagnostic) []pipeline.CheckpointDiagnostic {
if len(values) == 0 {
return nil
}
cloned := make([]pipeline.CheckpointDiagnostic, len(values))
for index, value := range values {
cloned[index] = pipeline.CheckpointDiagnostic{
Diagnostic: contracts.CloneProducerDiagnostics([]contracts.ProducerDiagnostic{value.Diagnostic})[0],
ValidatorKey: value.ValidatorKey,
}
}
return cloned
}
func artifactOutputDigests(outputs []pipeline.CheckpointArtifact) []pipeline.CheckpointFingerprint { func artifactOutputDigests(outputs []pipeline.CheckpointArtifact) []pipeline.CheckpointFingerprint {
values := make([]pipeline.CheckpointFingerprint, 0, len(outputs)) values := make([]pipeline.CheckpointFingerprint, 0, len(outputs))
for i, v := range outputs { for i, v := range outputs {
@@ -301,13 +313,12 @@ func artifactOutputDigests(outputs []pipeline.CheckpointArtifact) []pipeline.Che
return normalizeFingerprints(values) return normalizeFingerprints(values)
} }
func binaryEnvelopeFromContent(content []byte, mediaType string, metadata map[string]any, warnings []contracts.Warning) binaryEnvelope { func binaryEnvelopeFromContent(content []byte, mediaType string, metadata map[string]any) binaryEnvelope {
return binaryEnvelope{ return binaryEnvelope{
ContentBase64: base64.StdEncoding.EncodeToString(content), ContentBase64: base64.StdEncoding.EncodeToString(content),
ContentDigest: contentDigest(content), ContentDigest: contentDigest(content),
MediaType: mediaType, MediaType: mediaType,
Metadata: cloneMetadata(metadata), Metadata: cloneMetadata(metadata),
Warnings: cloneWarnings(warnings),
} }
} }
@@ -334,13 +345,6 @@ func cloneSourceUnits(units []source.SourceUnit) []source.SourceUnit {
return out return out
} }
func cloneWarnings(warnings []contracts.Warning) []contracts.Warning {
if len(warnings) == 0 {
return nil
}
return append([]contracts.Warning(nil), warnings...)
}
func cloneRejectedOutputs(rejected []contracts.RejectedOutput) []contracts.RejectedOutput { func cloneRejectedOutputs(rejected []contracts.RejectedOutput) []contracts.RejectedOutput {
if len(rejected) == 0 { if len(rejected) == 0 {
return nil return nil
@@ -461,13 +465,10 @@ func statusForRejected(rejected []contracts.RejectedOutput) StageStatus {
return StatusSucceeded return StatusSucceeded
} }
func validationStatusString(warnings []contracts.Warning, rejected []contracts.RejectedOutput) string { func validationStatusString(rejected []contracts.RejectedOutput) string {
if len(rejected) > 0 { if len(rejected) > 0 {
return "rejected" return "rejected"
} }
if len(warnings) > 0 {
return "approved_with_warnings"
}
return "approved" return "approved"
} }

View File

@@ -17,7 +17,7 @@ func TestRootBasedRecorderOutputIsReusable(t *testing.T) {
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
if err := recorder.ExtractSucceeded("lane", "module", nil, nil, nil, nil); err != nil { if err := recorder.ExtractSucceeded("lane", "module", nil, nil, nil); err != nil {
t.Fatal(err) t.Fatal(err)
} }
loader, err := NewFilesystemLoader(root, identity) loader, err := NewFilesystemLoader(root, identity)
@@ -44,7 +44,7 @@ func TestStepAwareRecorderAndLoaderIsolateLaneState(t *testing.T) {
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
if err := recorder.(pipeline.StepCheckpointRecorder).ExtractSucceededForStep("step-a", "lane", "module", nil, nil, nil, nil); err != nil { if err := recorder.(pipeline.StepCheckpointRecorder).ExtractSucceededForStep("step-a", "lane", "module", nil, nil, nil); err != nil {
t.Fatal(err) t.Fatal(err)
} }
loader, err := NewFilesystemLoader(root, identity) loader, err := NewFilesystemLoader(root, identity)
@@ -87,7 +87,7 @@ func TestStepAwareCheckpointPreservesDistinctDotIdentities(t *testing.T) {
LaneID: "lane", ModuleKey: "normalize-module", SourceID: "source", ChunkID: "chunk", ChunkRef: source.SourceRef{SourceID: "source", StartUnitID: 1, EndUnitID: 1}, SchemaDigest: "sha256:schema", LaneID: "lane", ModuleKey: "normalize-module", SourceID: "source", ChunkID: "chunk", ChunkRef: source.SourceRef{SourceID: "source", StartUnitID: 1, EndUnitID: 1}, SchemaDigest: "sha256:schema",
Artifact: contracts.SerializedArtifact{Kind: "kind", Schema: contracts.ArtifactSchema{ID: "schema", Name: "Schema", Version: "1"}, MediaType: "application/json", Content: []byte(test.content)}, Artifact: contracts.SerializedArtifact{Kind: "kind", Schema: contracts.ArtifactSchema{ID: "schema", Name: "Schema", Version: "1"}, MediaType: "application/json", Content: []byte(test.content)},
} }
if err := stepRecorder.NormalizeSucceededForStep(test.stepID, "lane", "normalize-module", nil, artifact, nil); err != nil { if err := stepRecorder.NormalizeSucceededForStep(test.stepID, "lane", "normalize-module", nil, artifact); err != nil {
t.Fatalf("record %q: %v", test.stepID, err) t.Fatalf("record %q: %v", test.stepID, err)
} }
} }
@@ -119,7 +119,7 @@ func TestStepAwareCheckpointPreservesDistinctDotIdentities(t *testing.T) {
} }
func TestCheckpointSchemaCompatibilityIdentifiers(t *testing.T) { func TestCheckpointSchemaCompatibilityIdentifiers(t *testing.T) {
if WorkspaceSchemaVersion != "notarius.workspace.v3" || WorkspaceSchemaVersionV2 != "notarius.workspace.v2" || WorkspaceSchemaVersionV1 != "notarius.workspace.v1" { if WorkspaceSchemaVersion != "notarius.workspace.v4" || WorkspaceSchemaVersionV3 != "notarius.workspace.v3" || WorkspaceSchemaVersionV2 != "notarius.workspace.v2" || WorkspaceSchemaVersionV1 != "notarius.workspace.v1" {
t.Fatal("checkpoint schema identifiers are incorrect") t.Fatal("checkpoint schema identifiers are incorrect")
} }
} }

View File

@@ -14,6 +14,7 @@ import (
"syscall" "syscall"
"gitea.maximumdirect.net/eric/notarius/internal/core/source" "gitea.maximumdirect.net/eric/notarius/internal/core/source"
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline" "gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
) )
@@ -340,6 +341,9 @@ func validateRecord(record pipeline.ChunkPlanRecord, requestedDigest string) err
if record.SchemaVersion != SchemaVersion { if record.SchemaVersion != SchemaVersion {
return fmt.Errorf("schema_version %q is not supported", record.SchemaVersion) return fmt.Errorf("schema_version %q is not supported", record.SchemaVersion)
} }
if err := contracts.ValidateProducerDiagnostics(record.Diagnostics); err != nil {
return fmt.Errorf("diagnostics: %w", err)
}
if _, err := digestPathSegment(requestedDigest); err != nil { if _, err := digestPathSegment(requestedDigest); err != nil {
return err return err
} }

View File

@@ -281,7 +281,7 @@ func TestFilesystemStoreReportsInvalidRecordsAsRecoverable(t *testing.T) {
return bytes.Replace(data, []byte(`{"schema_version"`), []byte(`{"SENTINEL_UNKNOWN_FIELD":true,"schema_version"`), 1) return bytes.Replace(data, []byte(`{"schema_version"`), []byte(`{"SENTINEL_UNKNOWN_FIELD":true,"schema_version"`), 1)
}}, }},
{name: "truncated JSON", mutate: func(data []byte) []byte { return data[:len(data)/2] }}, {name: "truncated JSON", mutate: func(data []byte) []byte { return data[:len(data)/2] }},
{name: "legacy v1 record", mutate: replaceJSON(`notarius.chunk-plan.v2`, `notarius.chunk-plan.v1`)}, {name: "legacy v2 record", mutate: replaceJSON(`notarius.chunk-plan.v3`, `notarius.chunk-plan.v2`)},
{name: "source mismatch", mutate: replaceJSON(testSourceDigest, "sha256:"+strings.Repeat("b", 64))}, {name: "source mismatch", mutate: replaceJSON(testSourceDigest, "sha256:"+strings.Repeat("b", 64))},
{name: "plan digest mismatch", mutate: func(data []byte) []byte { {name: "plan digest mismatch", mutate: func(data []byte) []byte {
prefix := []byte(`"plan_digest":"sha256:`) prefix := []byte(`"plan_digest":"sha256:`)
@@ -498,7 +498,13 @@ func testRecord(t *testing.T, value int) pipeline.ChunkPlanRecord {
References: []artifacts.ReferenceProvenance{{Stage: "chunk", SlotName: "guide", OriginType: "file", OriginURI: "file:///guide.txt", Digest: "sha256:reference"}}, References: []artifacts.ReferenceProvenance{{Stage: "chunk", SlotName: "guide", OriginType: "file", OriginURI: "file:///guide.txt", Digest: "sha256:reference"}},
Metadata: map[string]any{"prompt_id": "test/prompt", "enabled": true}, Metadata: map[string]any{"prompt_id": "test/prompt", "enabled": true},
}, },
Warnings: []contracts.Warning{{Scope: "chunk/test", ReasonCode: "observed", Message: "warning"}}, Diagnostics: []contracts.ProducerDiagnostic{{
Disposition: contracts.DiagnosticDispositionWarning,
Category: contracts.DiagnosticCategoryConfiguration,
ReasonCode: "empty_reference",
OccurrenceCount: 1,
Samples: []contracts.DiagnosticSample{{Scope: "reference", Message: "Reference was empty."}},
}},
CreatedAt: time.Date(2026, 7, 18, 12, 0, 0, 0, time.UTC), CreatedAt: time.Date(2026, 7, 18, 12, 0, 0, 0, time.UTC),
} }
} }

View File

@@ -163,7 +163,7 @@ type ChunkRequest struct {
type ChunkPlanResult struct { type ChunkPlanResult struct {
Plan source.ChunkPlan `json:"plan"` Plan source.ChunkPlan `json:"plan"`
Warnings []Warning `json:"warnings,omitempty"` Diagnostics []ProducerDiagnostic `json:"diagnostics,omitempty"`
ModelCandidate *ModelCandidate `json:"-"` ModelCandidate *ModelCandidate `json:"-"`
} }
@@ -288,20 +288,14 @@ type ValidationResult struct {
Message string `json:"message,omitempty"` Message string `json:"message,omitempty"`
CorrectionGuidance string `json:"-"` CorrectionGuidance string `json:"-"`
DiagnosticArtifactPath string `json:"diagnostic_artifact_path,omitempty"` DiagnosticArtifactPath string `json:"diagnostic_artifact_path,omitempty"`
Warnings []Warning `json:"warnings,omitempty"` Diagnostics []ProducerDiagnostic `json:"diagnostics,omitempty"`
}
type Warning struct {
Scope string `json:"scope,omitempty"`
ReasonCode string `json:"reason_code"`
Message string `json:"message"`
} }
type OutputRequest struct { type OutputRequest struct {
Manifest artifacts.RunManifest `json:"manifest"` Manifest artifacts.RunManifest `json:"manifest"`
NormalizeOutputs []SerializedOutput `json:"normalize_outputs,omitempty"` NormalizeOutputs []SerializedOutput `json:"normalize_outputs,omitempty"`
Rejected []RejectedOutput `json:"rejected,omitempty"` Rejected []RejectedOutput `json:"rejected,omitempty"`
Warnings []Warning `json:"warnings,omitempty"` Diagnostics DiagnosticCollection `json:"diagnostics,omitempty"`
LLMProfile string `json:"llm_profile,omitempty"` LLMProfile string `json:"llm_profile,omitempty"`
StructuredOutputRepairAttempts *int `json:"structured_output_repair_attempts,omitempty"` StructuredOutputRepairAttempts *int `json:"structured_output_repair_attempts,omitempty"`
Metadata map[string]any `json:"metadata,omitempty"` Metadata map[string]any `json:"metadata,omitempty"`
@@ -317,7 +311,6 @@ type OutputFile struct {
type OutputResult struct { type OutputResult struct {
Files []OutputFile `json:"files,omitempty"` Files []OutputFile `json:"files,omitempty"`
Warnings []Warning `json:"warnings,omitempty"`
} }
type OutputEncoder interface { type OutputEncoder interface {

View File

@@ -126,6 +126,9 @@ func (candidate ModelCandidate) Validate() error {
} }
func ValidateValidationResult(result ValidationResult) error { func ValidateValidationResult(result ValidationResult) error {
if err := ValidateProducerDiagnostics(result.Diagnostics); err != nil {
return fmt.Errorf("validation diagnostics: %w", err)
}
if !result.Approved && result.ReasonCode == "" { if !result.Approved && result.ReasonCode == "" {
return errors.New("validation rejection reason code must not be empty") return errors.New("validation rejection reason code must not be empty")
} }

View File

@@ -75,6 +75,9 @@ func TestCorrectionContractsRejectInvalidContent(t *testing.T) {
{"oversized correction guidance", func() error { {"oversized correction guidance", func() error {
return ValidateValidationResult(ValidationResult{ReasonCode: "invalid", CorrectionGuidance: tooLongValidationGuidance}) return ValidateValidationResult(ValidationResult{ReasonCode: "invalid", CorrectionGuidance: tooLongValidationGuidance})
}}, }},
{"invalid diagnostics", func() error {
return ValidateValidationResult(ValidationResult{Approved: true, Diagnostics: []ProducerDiagnostic{{}}})
}},
} { } {
t.Run(test.name, func(t *testing.T) { t.Run(test.name, func(t *testing.T) {
if err := test.call(); err == nil { if err := test.call(); err == nil {

View File

@@ -0,0 +1,401 @@
package contracts
import (
"errors"
"fmt"
"strings"
"unicode/utf8"
)
const (
MaxDiagnosticReasonCodeBytes = 128
MaxDiagnosticScopeBytes = 512
MaxDiagnosticMessageBytes = 4 * 1024
MaxDiagnosticSamples = 3
MaxProducerDiagnosticGroups = 64
)
// DiagnosticDisposition identifies the operator significance of a producer
// finding. Warnings are reserved for process-level degradation or incomplete
// configured work.
type DiagnosticDisposition string
const (
DiagnosticDispositionWarning DiagnosticDisposition = "warning"
DiagnosticDispositionAdvisory DiagnosticDisposition = "advisory"
DiagnosticDispositionObservation DiagnosticDisposition = "observation"
)
// DiagnosticCategory gives a stable, bounded classification for a producer
// finding.
type DiagnosticCategory string
const (
DiagnosticCategoryConfiguration DiagnosticCategory = "configuration"
DiagnosticCategoryDegradation DiagnosticCategory = "degradation"
DiagnosticCategoryValidationIncomplete DiagnosticCategory = "validation_incomplete"
DiagnosticCategoryFallback DiagnosticCategory = "fallback"
DiagnosticCategoryDataQuality DiagnosticCategory = "data_quality"
DiagnosticCategoryNormalization DiagnosticCategory = "normalization"
)
// DiagnosticOriginStage identifies the framework operation that promoted a
// diagnostic. It is framework-owned rather than producer-owned.
type DiagnosticOriginStage string
const (
DiagnosticOriginStageReferences DiagnosticOriginStage = "references"
DiagnosticOriginStageChunk DiagnosticOriginStage = "chunk"
DiagnosticOriginStageExtract DiagnosticOriginStage = "extract"
DiagnosticOriginStageMerge DiagnosticOriginStage = "merge"
DiagnosticOriginStageNormalize DiagnosticOriginStage = "normalize"
)
// DiagnosticSample is a bounded, safe example of a diagnostic occurrence.
// Chunk identity is attached by the framework when it promotes a producer
// diagnostic into a final group.
type DiagnosticSample struct {
Scope string `json:"scope"`
Message string `json:"message"`
ChunkID string `json:"chunk_id,omitempty"`
ChunkIndex *int `json:"chunk_index,omitempty"`
}
// ProducerDiagnostic is the locally grouped form returned by one producer or
// validator. It intentionally has no pipeline origin.
type ProducerDiagnostic struct {
Disposition DiagnosticDisposition `json:"disposition"`
Category DiagnosticCategory `json:"category"`
ReasonCode string `json:"reason_code"`
OccurrenceCount int `json:"occurrence_count"`
Samples []DiagnosticSample `json:"samples"`
OmittedSampleCount int `json:"omitted_sample_count"`
}
// DiagnosticOrigin is framework-owned context used to distinguish findings
// from different pipeline locations during final aggregation.
type DiagnosticOrigin struct {
Stage DiagnosticOriginStage `json:"stage"`
StepID string `json:"step_id,omitempty"`
LaneID string `json:"lane_id,omitempty"`
ModuleKey string `json:"module_key,omitempty"`
ValidatorKey string `json:"validator_key,omitempty"`
}
// DiagnosticGroup is a producer diagnostic after framework origin enrichment.
type DiagnosticGroup struct {
Disposition DiagnosticDisposition `json:"disposition"`
Category DiagnosticCategory `json:"category"`
ReasonCode string `json:"reason_code"`
Origin DiagnosticOrigin `json:"origin"`
OccurrenceCount int `json:"occurrence_count"`
Samples []DiagnosticSample `json:"samples"`
OmittedSampleCount int `json:"omitted_sample_count"`
}
// DiagnosticCollection is the grouped collection supplied to later durable
// and presentation boundaries. Global aggregation policy is applied by the
// framework before it reaches those boundaries.
type DiagnosticCollection struct {
Groups []DiagnosticGroup `json:"groups"`
Truncated bool `json:"truncated"`
UnrepresentedOccurrenceCount int `json:"unrepresented_occurrence_count"`
}
// DiagnosticProjection is the validated warning/non-warning view used by
// durable and presentation boundaries. Occurrence totals are checked before
// they leave the framework contract.
type DiagnosticProjection struct {
Warnings []DiagnosticGroup
Diagnostics []DiagnosticGroup
WarningOccurrenceCount int
DiagnosticOccurrenceCount int
}
// Validate checks a producer-local diagnostic against the public safety and
// classification contract.
func (diagnostic ProducerDiagnostic) Validate() error {
return validateDiagnostic(
diagnostic.Disposition,
diagnostic.Category,
diagnostic.ReasonCode,
diagnostic.OccurrenceCount,
diagnostic.Samples,
diagnostic.OmittedSampleCount,
false,
)
}
// ValidateProducerDiagnostics validates the complete set returned by one
// producer or validator result.
func ValidateProducerDiagnostics(diagnostics []ProducerDiagnostic) error {
if len(diagnostics) > MaxProducerDiagnosticGroups {
return errors.New("producer diagnostics exceed maximum group count")
}
for index, diagnostic := range diagnostics {
if err := diagnostic.Validate(); err != nil {
return fmt.Errorf("producer diagnostic %d: %w", index, err)
}
}
return nil
}
// Validate checks framework-owned origin fields.
func (origin DiagnosticOrigin) Validate() error {
switch origin.Stage {
case DiagnosticOriginStageReferences, DiagnosticOriginStageChunk, DiagnosticOriginStageExtract, DiagnosticOriginStageMerge, DiagnosticOriginStageNormalize:
default:
return errors.New("diagnostic origin stage is invalid")
}
for _, field := range []struct {
name string
value string
}{
{name: "diagnostic origin step ID", value: origin.StepID},
{name: "diagnostic origin lane ID", value: origin.LaneID},
{name: "diagnostic origin module key", value: origin.ModuleKey},
{name: "diagnostic origin validator key", value: origin.ValidatorKey},
} {
if field.value != "" && (!utf8.ValidString(field.value) || strings.TrimSpace(field.value) == "") {
return fmt.Errorf("%s must be valid nonblank UTF-8 when present", field.name)
}
}
return nil
}
// Validate checks a final origin-enriched group.
func (group DiagnosticGroup) Validate() error {
if err := group.Origin.Validate(); err != nil {
return err
}
return validateDiagnostic(
group.Disposition,
group.Category,
group.ReasonCode,
group.OccurrenceCount,
group.Samples,
group.OmittedSampleCount,
true,
)
}
// Validate checks the collection shape without imposing later global
// aggregation limits.
func (collection DiagnosticCollection) Validate() error {
if collection.UnrepresentedOccurrenceCount < 0 {
return errors.New("diagnostic collection unrepresented occurrence count must not be negative")
}
if !collection.Truncated && collection.UnrepresentedOccurrenceCount != 0 {
return errors.New("diagnostic collection has unrepresented occurrences without truncation")
}
seen := make(map[diagnosticGroupKey]struct{}, len(collection.Groups))
for index, group := range collection.Groups {
if err := group.Validate(); err != nil {
return fmt.Errorf("diagnostic group %d: %w", index, err)
}
key := diagnosticGroupKeyFromGroup(group)
if _, exists := seen[key]; exists {
return errors.New("diagnostic collection contains duplicate group identity")
}
seen[key] = struct{}{}
}
return nil
}
// ProjectDiagnosticCollection validates, partitions, and totals one finalized
// collection. Unrepresented occurrences belong to the non-warning projection.
func ProjectDiagnosticCollection(collection DiagnosticCollection) (DiagnosticProjection, error) {
if err := collection.Validate(); err != nil {
return DiagnosticProjection{}, err
}
projection := DiagnosticProjection{
Warnings: make([]DiagnosticGroup, 0),
Diagnostics: make([]DiagnosticGroup, 0),
}
for _, group := range collection.Groups {
if group.Disposition == DiagnosticDispositionWarning {
count, err := addDiagnosticOccurrences(projection.WarningOccurrenceCount, group.OccurrenceCount)
if err != nil {
return DiagnosticProjection{}, fmt.Errorf("warning occurrences: %w", err)
}
projection.WarningOccurrenceCount = count
projection.Warnings = append(projection.Warnings, cloneDiagnosticGroup(group))
continue
}
count, err := addDiagnosticOccurrences(projection.DiagnosticOccurrenceCount, group.OccurrenceCount)
if err != nil {
return DiagnosticProjection{}, fmt.Errorf("diagnostic occurrences: %w", err)
}
projection.DiagnosticOccurrenceCount = count
projection.Diagnostics = append(projection.Diagnostics, cloneDiagnosticGroup(group))
}
count, err := addDiagnosticOccurrences(projection.DiagnosticOccurrenceCount, collection.UnrepresentedOccurrenceCount)
if err != nil {
return DiagnosticProjection{}, fmt.Errorf("diagnostic occurrences: %w", err)
}
projection.DiagnosticOccurrenceCount = count
return projection, nil
}
// CloneProducerDiagnostics returns independent diagnostic slice ownership.
func CloneProducerDiagnostics(diagnostics []ProducerDiagnostic) []ProducerDiagnostic {
if len(diagnostics) == 0 {
return nil
}
cloned := make([]ProducerDiagnostic, len(diagnostics))
for index, diagnostic := range diagnostics {
cloned[index] = cloneProducerDiagnostic(diagnostic)
}
return cloned
}
// CloneDiagnosticCollection returns independent collection ownership.
func CloneDiagnosticCollection(collection DiagnosticCollection) DiagnosticCollection {
groups := collection.Groups
collection.Groups = make([]DiagnosticGroup, len(groups))
for index, group := range groups {
collection.Groups[index] = cloneDiagnosticGroup(group)
}
return collection
}
func validateDiagnostic(disposition DiagnosticDisposition, category DiagnosticCategory, reasonCode string, occurrenceCount int, samples []DiagnosticSample, omittedSampleCount int, allowChunkContext bool) error {
if !diagnosticCategoryAllowed(disposition, category) {
return errors.New("diagnostic disposition and category combination is invalid")
}
if err := validateDiagnosticText(reasonCode, MaxDiagnosticReasonCodeBytes, "diagnostic reason code"); err != nil {
return err
}
if occurrenceCount <= 0 {
return errors.New("diagnostic occurrence count must be positive")
}
if len(samples) == 0 {
return errors.New("diagnostic samples must not be empty")
}
if len(samples) > MaxDiagnosticSamples {
return errors.New("diagnostic samples exceed maximum count")
}
seen := make(map[diagnosticSampleKey]struct{}, len(samples))
for index, sample := range samples {
if err := validateDiagnosticSample(sample, allowChunkContext); err != nil {
return fmt.Errorf("diagnostic sample %d: %w", index, err)
}
key := diagnosticSampleKeyFromSample(sample)
if _, exists := seen[key]; exists {
return errors.New("diagnostic samples must be distinct")
}
seen[key] = struct{}{}
}
if occurrenceCount < len(samples) {
return errors.New("diagnostic occurrence count is smaller than sample count")
}
if omittedSampleCount != occurrenceCount-len(samples) {
return errors.New("diagnostic omitted sample count is inconsistent")
}
return nil
}
func diagnosticCategoryAllowed(disposition DiagnosticDisposition, category DiagnosticCategory) bool {
switch disposition {
case DiagnosticDispositionWarning:
return category == DiagnosticCategoryConfiguration || category == DiagnosticCategoryDegradation || category == DiagnosticCategoryValidationIncomplete || category == DiagnosticCategoryFallback
case DiagnosticDispositionAdvisory:
return category == DiagnosticCategoryDataQuality
case DiagnosticDispositionObservation:
return category == DiagnosticCategoryNormalization
default:
return false
}
}
func validateDiagnosticSample(sample DiagnosticSample, allowChunkContext bool) error {
if err := validateDiagnosticText(sample.Scope, MaxDiagnosticScopeBytes, "diagnostic sample scope"); err != nil {
return err
}
if err := validateDiagnosticText(sample.Message, MaxDiagnosticMessageBytes, "diagnostic sample message"); err != nil {
return err
}
if !allowChunkContext && (sample.ChunkID != "" || sample.ChunkIndex != nil) {
return errors.New("producer diagnostic sample must not include framework chunk context")
}
if sample.ChunkID != "" && (!utf8.ValidString(sample.ChunkID) || strings.TrimSpace(sample.ChunkID) == "") {
return errors.New("diagnostic sample chunk ID must be valid nonblank UTF-8 when present")
}
if sample.ChunkIndex != nil && *sample.ChunkIndex < 0 {
return errors.New("diagnostic sample chunk index must not be negative")
}
return nil
}
func validateDiagnosticText(value string, maximum int, name string) error {
if !utf8.ValidString(value) {
return fmt.Errorf("%s must be valid UTF-8", name)
}
if strings.TrimSpace(value) == "" {
return fmt.Errorf("%s must not be blank", name)
}
if len(value) > maximum {
return fmt.Errorf("%s exceeds maximum length", name)
}
return nil
}
func cloneProducerDiagnostic(diagnostic ProducerDiagnostic) ProducerDiagnostic {
diagnostic.Samples = cloneDiagnosticSamples(diagnostic.Samples)
return diagnostic
}
func cloneDiagnosticGroup(group DiagnosticGroup) DiagnosticGroup {
group.Samples = cloneDiagnosticSamples(group.Samples)
return group
}
func addDiagnosticOccurrences(current, incoming int) (int, error) {
if incoming > int(^uint(0)>>1)-current {
return 0, errors.New("occurrence count overflow")
}
return current + incoming, nil
}
func cloneDiagnosticSamples(samples []DiagnosticSample) []DiagnosticSample {
if len(samples) == 0 {
return nil
}
cloned := make([]DiagnosticSample, len(samples))
for index, sample := range samples {
if sample.ChunkIndex != nil {
chunkIndex := *sample.ChunkIndex
sample.ChunkIndex = &chunkIndex
}
cloned[index] = sample
}
return cloned
}
type diagnosticSampleKey struct {
scope string
message string
chunkID string
chunkIndex int
hasChunkIndex bool
}
func diagnosticSampleKeyFromSample(sample DiagnosticSample) diagnosticSampleKey {
key := diagnosticSampleKey{scope: sample.Scope, message: sample.Message, chunkID: sample.ChunkID}
if sample.ChunkIndex != nil {
key.chunkIndex = *sample.ChunkIndex
key.hasChunkIndex = true
}
return key
}
type diagnosticGroupKey struct {
disposition DiagnosticDisposition
category DiagnosticCategory
reasonCode string
origin DiagnosticOrigin
}
func diagnosticGroupKeyFromGroup(group DiagnosticGroup) diagnosticGroupKey {
return diagnosticGroupKey{disposition: group.Disposition, category: group.Category, reasonCode: group.ReasonCode, origin: group.Origin}
}

View File

@@ -0,0 +1,198 @@
package contracts
import (
"strings"
"testing"
)
func TestProducerDiagnosticValidationAcceptsClassificationMatrix(t *testing.T) {
for _, test := range []struct {
name string
disposition DiagnosticDisposition
category DiagnosticCategory
}{
{name: "configuration warning", disposition: DiagnosticDispositionWarning, category: DiagnosticCategoryConfiguration},
{name: "degradation warning", disposition: DiagnosticDispositionWarning, category: DiagnosticCategoryDegradation},
{name: "incomplete validation warning", disposition: DiagnosticDispositionWarning, category: DiagnosticCategoryValidationIncomplete},
{name: "fallback warning", disposition: DiagnosticDispositionWarning, category: DiagnosticCategoryFallback},
{name: "quality advisory", disposition: DiagnosticDispositionAdvisory, category: DiagnosticCategoryDataQuality},
{name: "normalization observation", disposition: DiagnosticDispositionObservation, category: DiagnosticCategoryNormalization},
} {
t.Run(test.name, func(t *testing.T) {
diagnostic := validProducerDiagnostic()
diagnostic.Disposition = test.disposition
diagnostic.Category = test.category
if err := diagnostic.Validate(); err != nil {
t.Fatalf("Validate() error = %v", err)
}
})
}
}
func TestProducerDiagnosticValidationRejectsInvalidFieldsAndCounts(t *testing.T) {
tooLongReason := strings.Repeat("r", MaxDiagnosticReasonCodeBytes+1)
tooLongScope := strings.Repeat("s", MaxDiagnosticScopeBytes+1)
tooLongMessage := strings.Repeat("m", MaxDiagnosticMessageBytes+1)
for _, test := range []struct {
name string
mutate func(*ProducerDiagnostic)
}{
{name: "invalid classification", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.Category = DiagnosticCategoryDataQuality }},
{name: "blank reason", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.ReasonCode = " \t" }},
{name: "invalid reason UTF-8", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.ReasonCode = string([]byte{0xff}) }},
{name: "oversized reason", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.ReasonCode = tooLongReason }},
{name: "blank scope", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.Samples[0].Scope = "\n" }},
{name: "invalid scope UTF-8", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.Samples[0].Scope = string([]byte{0xff}) }},
{name: "oversized scope", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.Samples[0].Scope = tooLongScope }},
{name: "blank message", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.Samples[0].Message = " " }},
{name: "invalid message UTF-8", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.Samples[0].Message = string([]byte{0xff}) }},
{name: "oversized message", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.Samples[0].Message = tooLongMessage }},
{name: "zero occurrences", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.OccurrenceCount = 0 }},
{name: "missing samples", mutate: func(diagnostic *ProducerDiagnostic) {
diagnostic.Samples = nil
diagnostic.OmittedSampleCount = diagnostic.OccurrenceCount
}},
{name: "too many samples", mutate: func(diagnostic *ProducerDiagnostic) {
diagnostic.OccurrenceCount = 4
diagnostic.Samples = []DiagnosticSample{{Scope: "one", Message: "one"}, {Scope: "two", Message: "two"}, {Scope: "three", Message: "three"}, {Scope: "four", Message: "four"}}
diagnostic.OmittedSampleCount = 0
}},
{name: "duplicate samples", mutate: func(diagnostic *ProducerDiagnostic) {
diagnostic.OccurrenceCount = 2
diagnostic.Samples = []DiagnosticSample{{Scope: "scope", Message: "message"}, {Scope: "scope", Message: "message"}}
diagnostic.OmittedSampleCount = 0
}},
{name: "inconsistent omission", mutate: func(diagnostic *ProducerDiagnostic) { diagnostic.OmittedSampleCount = 1 }},
{name: "producer chunk context", mutate: func(diagnostic *ProducerDiagnostic) { chunkIndex := 0; diagnostic.Samples[0].ChunkIndex = &chunkIndex }},
} {
t.Run(test.name, func(t *testing.T) {
diagnostic := validProducerDiagnostic()
test.mutate(&diagnostic)
if err := diagnostic.Validate(); err == nil {
t.Fatal("Validate() error = nil, want invalid diagnostic error")
}
})
}
}
func TestValidateProducerDiagnosticsEnforcesLocalGroupBound(t *testing.T) {
diagnostics := make([]ProducerDiagnostic, MaxProducerDiagnosticGroups)
for index := range diagnostics {
diagnostics[index] = validProducerDiagnostic()
diagnostics[index].ReasonCode = "reason-" + string(rune('a'+index))
}
if err := ValidateProducerDiagnostics(diagnostics); err != nil {
t.Fatalf("ValidateProducerDiagnostics() error = %v", err)
}
diagnostics = append(diagnostics, validProducerDiagnostic())
if err := ValidateProducerDiagnostics(diagnostics); err == nil {
t.Fatal("ValidateProducerDiagnostics() error = nil, want excessive-group error")
}
}
func TestDiagnosticGroupValidationPreservesChunkIndexZero(t *testing.T) {
chunkIndex := 0
group := DiagnosticGroup{
Disposition: DiagnosticDispositionAdvisory,
Category: DiagnosticCategoryDataQuality,
ReasonCode: "unresolved",
Origin: DiagnosticOrigin{Stage: DiagnosticOriginStageExtract, StepID: "extract", LaneID: "spells", ModuleKey: "dnd/spells", ValidatorKey: "dnd/spells/source-relatedness"},
OccurrenceCount: 1,
Samples: []DiagnosticSample{{Scope: "spells[0]", Message: "Spell was not found", ChunkID: "chunk-1", ChunkIndex: &chunkIndex}},
}
if err := group.Validate(); err != nil {
t.Fatalf("Validate() error = %v", err)
}
collection := DiagnosticCollection{Groups: []DiagnosticGroup{group}}
if err := collection.Validate(); err != nil {
t.Fatalf("DiagnosticCollection.Validate() error = %v", err)
}
}
func TestDiagnosticGroupRejectsRepeatedSampleWithEqualChunkIndex(t *testing.T) {
firstIndex := 0
secondIndex := 0
group := DiagnosticGroup{
Disposition: DiagnosticDispositionAdvisory,
Category: DiagnosticCategoryDataQuality,
ReasonCode: "unresolved",
Origin: DiagnosticOrigin{Stage: DiagnosticOriginStageExtract, StepID: "extract", LaneID: "spells", ModuleKey: "dnd/spells"},
OccurrenceCount: 2,
Samples: []DiagnosticSample{
{Scope: "spells[0]", Message: "Spell was not found", ChunkID: "chunk-1", ChunkIndex: &firstIndex},
{Scope: "spells[0]", Message: "Spell was not found", ChunkID: "chunk-1", ChunkIndex: &secondIndex},
},
}
if err := group.Validate(); err == nil {
t.Fatal("Validate() error = nil, want duplicate sample error")
}
}
func TestCloneDiagnosticCollectionOwnsGroupsAndChunkIndex(t *testing.T) {
chunkIndex := 0
collection := DiagnosticCollection{Groups: []DiagnosticGroup{{
Disposition: DiagnosticDispositionAdvisory,
Category: DiagnosticCategoryDataQuality,
ReasonCode: "unresolved",
Origin: DiagnosticOrigin{Stage: DiagnosticOriginStageExtract, StepID: "extract", LaneID: "spells", ModuleKey: "dnd/spells"},
OccurrenceCount: 1,
Samples: []DiagnosticSample{{Scope: "spells[0]", Message: "Spell was not found", ChunkIndex: &chunkIndex}},
}}}
cloned := CloneDiagnosticCollection(collection)
collection.Groups[0].Samples[0].Message = "changed"
*collection.Groups[0].Samples[0].ChunkIndex = 1
if got := cloned.Groups[0].Samples[0]; got.Message != "Spell was not found" || got.ChunkIndex == nil || *got.ChunkIndex != 0 {
t.Fatalf("cloned sample = %#v, want independently owned original", got)
}
}
func TestProjectDiagnosticCollectionPartitionsAndChecksTotals(t *testing.T) {
warning := validDiagnosticGroup(DiagnosticDispositionWarning, DiagnosticCategoryFallback, "fallback", 2)
diagnostic := validDiagnosticGroup(DiagnosticDispositionAdvisory, DiagnosticCategoryDataQuality, "quality", 3)
collection := DiagnosticCollection{
Groups: []DiagnosticGroup{warning, diagnostic},
Truncated: true,
UnrepresentedOccurrenceCount: 4,
}
projection, err := ProjectDiagnosticCollection(collection)
if err != nil {
t.Fatal(err)
}
if len(projection.Warnings) != 1 || len(projection.Diagnostics) != 1 || projection.WarningOccurrenceCount != 2 || projection.DiagnosticOccurrenceCount != 7 {
t.Fatalf("projection = %#v, want partitioned exact totals", projection)
}
collection.Groups[0].Samples[0].Message = "mutated"
if projection.Warnings[0].Samples[0].Message != "message" {
t.Fatal("projection retained caller-owned sample storage")
}
overflow := DiagnosticCollection{Groups: []DiagnosticGroup{
validDiagnosticGroup(DiagnosticDispositionWarning, DiagnosticCategoryFallback, "first", int(^uint(0)>>1)),
validDiagnosticGroup(DiagnosticDispositionWarning, DiagnosticCategoryFallback, "second", 1),
}}
if _, err := ProjectDiagnosticCollection(overflow); err == nil {
t.Fatal("ProjectDiagnosticCollection() overflow error = nil")
}
}
func validDiagnosticGroup(disposition DiagnosticDisposition, category DiagnosticCategory, reason string, occurrences int) DiagnosticGroup {
return DiagnosticGroup{
Disposition: disposition,
Category: category,
ReasonCode: reason,
Origin: DiagnosticOrigin{Stage: DiagnosticOriginStageNormalize, StepID: "step", LaneID: "lane", ModuleKey: "module"},
OccurrenceCount: occurrences,
Samples: []DiagnosticSample{{Scope: "scope", Message: "message"}},
OmittedSampleCount: occurrences - 1,
}
}
func validProducerDiagnostic() ProducerDiagnostic {
return ProducerDiagnostic{
Disposition: DiagnosticDispositionWarning,
Category: DiagnosticCategoryConfiguration,
ReasonCode: "empty_reference",
OccurrenceCount: 1,
Samples: []DiagnosticSample{{Scope: "references.glossary", Message: "Reference is empty"}},
}
}

View File

@@ -48,7 +48,7 @@ type TypedExtractionRequest struct {
type TypedExtractionResult[T any] struct { type TypedExtractionResult[T any] struct {
Value T Value T
Warnings []Warning Diagnostics []ProducerDiagnostic
ModelCandidate *ModelCandidate ModelCandidate *ModelCandidate
} }
@@ -73,7 +73,7 @@ type TypedMergeRequest[T any] struct {
type TypedMergeResult[T any] struct { type TypedMergeResult[T any] struct {
Value T Value T
Warnings []Warning Diagnostics []ProducerDiagnostic
ModelCandidate *ModelCandidate ModelCandidate *ModelCandidate
} }
@@ -97,16 +97,17 @@ type TypedNormalizeRequest[T any] struct {
type TypedNormalizeResult[T any] struct { type TypedNormalizeResult[T any] struct {
Value T Value T
Warnings []Warning Diagnostics []ProducerDiagnostic
Retry *NormalizeRetry Retry *NormalizeRetry
ModelCandidate *ModelCandidate ModelCandidate *ModelCandidate
} }
// Normalize retry diagnostic limits bound module-provided values before the // Normalize retry limits bound module-provided control and diagnostic text
// framework persists them in debug artifacts. // before the framework consumes or records it.
const ( const (
MaxNormalizeRetryReasonCodeBytes = 128 MaxNormalizeRetryReasonCodeBytes = 128
MaxNormalizeRetryMessageBytes = 4096 MaxNormalizeRetryMessageBytes = 4096
MaxNormalizeRetryCorrectionGuidanceBytes = 4096
) )
// NormalizeRetry asks the framework to retry normalization while retaining a // NormalizeRetry asks the framework to retry normalization while retaining a
@@ -114,7 +115,8 @@ const (
type NormalizeRetry struct { type NormalizeRetry struct {
ReasonCode string ReasonCode string
Message string Message string
FallbackWarnings []Warning CorrectionGuidance string
FallbackDiagnostics []ProducerDiagnostic
} }
type Normalizer[T any] interface { type Normalizer[T any] interface {

View File

@@ -0,0 +1,164 @@
package diagnostics
import (
"errors"
"fmt"
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
)
const (
MaxWarningGroups = 128
MaxNonWarningGroups = 256
)
// Aggregator merges origin-enriched diagnostics in caller-supplied canonical
// order. Its zero value is ready for use.
type Aggregator struct {
groups []contracts.DiagnosticGroup
indices map[groupKey]int
warningGroups int
nonWarningGroups int
warningOccurrences int
nonWarningOccurrences int
unrepresentedOccurrences int
}
// Add validates and incorporates one final diagnostic group. Actionable
// warnings cannot overflow; later non-warning groups are represented by exact
// unrepresented-occurrence metadata once their fixed bound is reached.
func (aggregator *Aggregator) Add(group contracts.DiagnosticGroup) error {
if err := group.Validate(); err != nil {
return fmt.Errorf("diagnostic group: %w", err)
}
if aggregator.indices == nil {
aggregator.indices = make(map[groupKey]int)
}
key := groupKeyFromGroup(group)
if index, exists := aggregator.indices[key]; exists {
if err := aggregator.checkOccurrenceTotal(group.Disposition, group.OccurrenceCount); err != nil {
return err
}
if err := aggregator.merge(index, group); err != nil {
return err
}
aggregator.addOccurrenceTotal(group.Disposition, group.OccurrenceCount)
return nil
}
if group.Disposition == contracts.DiagnosticDispositionWarning {
if aggregator.warningGroups >= MaxWarningGroups {
return errors.New("diagnostic warning groups exceed maximum count")
}
} else if aggregator.nonWarningGroups >= MaxNonWarningGroups {
if err := aggregator.checkOccurrenceTotal(group.Disposition, group.OccurrenceCount); err != nil {
return err
}
aggregator.addOccurrenceTotal(group.Disposition, group.OccurrenceCount)
return aggregator.addUnrepresented(group.OccurrenceCount)
}
if err := aggregator.checkOccurrenceTotal(group.Disposition, group.OccurrenceCount); err != nil {
return err
}
if group.Disposition == contracts.DiagnosticDispositionWarning {
aggregator.warningGroups++
} else {
aggregator.nonWarningGroups++
}
aggregator.addOccurrenceTotal(group.Disposition, group.OccurrenceCount)
aggregator.indices[key] = len(aggregator.groups)
aggregator.groups = append(aggregator.groups, contracts.CloneDiagnosticCollection(contracts.DiagnosticCollection{Groups: []contracts.DiagnosticGroup{group}}).Groups[0])
return nil
}
func (aggregator *Aggregator) checkOccurrenceTotal(disposition contracts.DiagnosticDisposition, count int) error {
current := aggregator.nonWarningOccurrences
if disposition == contracts.DiagnosticDispositionWarning {
current = aggregator.warningOccurrences
}
if count > maximumInt()-current {
return errors.New("diagnostic occurrence count overflow")
}
return nil
}
func (aggregator *Aggregator) addOccurrenceTotal(disposition contracts.DiagnosticDisposition, count int) {
if disposition == contracts.DiagnosticDispositionWarning {
aggregator.warningOccurrences += count
return
}
aggregator.nonWarningOccurrences += count
}
// Collection returns an independently owned grouped result in first-occurrence
// order.
func (aggregator *Aggregator) Collection() contracts.DiagnosticCollection {
if aggregator == nil {
return contracts.DiagnosticCollection{}
}
return contracts.CloneDiagnosticCollection(contracts.DiagnosticCollection{
Groups: aggregator.groups,
Truncated: aggregator.unrepresentedOccurrences > 0,
UnrepresentedOccurrenceCount: aggregator.unrepresentedOccurrences,
})
}
func (aggregator *Aggregator) merge(index int, incoming contracts.DiagnosticGroup) error {
current := &aggregator.groups[index]
if incoming.OccurrenceCount > maximumInt()-current.OccurrenceCount {
return errors.New("diagnostic occurrence count overflow")
}
current.OccurrenceCount += incoming.OccurrenceCount
for _, sample := range incoming.Samples {
if len(current.Samples) == contracts.MaxDiagnosticSamples || containsGroupSample(current.Samples, sample) {
continue
}
current.Samples = append(current.Samples, cloneGroupSample(sample))
}
current.OmittedSampleCount = current.OccurrenceCount - len(current.Samples)
return nil
}
func (aggregator *Aggregator) addUnrepresented(count int) error {
if count > maximumInt()-aggregator.unrepresentedOccurrences {
return errors.New("diagnostic unrepresented occurrence count overflow")
}
aggregator.unrepresentedOccurrences += count
return nil
}
func containsGroupSample(samples []contracts.DiagnosticSample, candidate contracts.DiagnosticSample) bool {
for _, sample := range samples {
if sample.Scope != candidate.Scope || sample.Message != candidate.Message || sample.ChunkID != candidate.ChunkID {
continue
}
if sample.ChunkIndex == nil || candidate.ChunkIndex == nil {
if sample.ChunkIndex == candidate.ChunkIndex {
return true
}
continue
}
if *sample.ChunkIndex == *candidate.ChunkIndex {
return true
}
}
return false
}
func cloneGroupSample(sample contracts.DiagnosticSample) contracts.DiagnosticSample {
if sample.ChunkIndex != nil {
value := *sample.ChunkIndex
sample.ChunkIndex = &value
}
return sample
}
type groupKey struct {
disposition contracts.DiagnosticDisposition
category contracts.DiagnosticCategory
reasonCode string
origin contracts.DiagnosticOrigin
}
func groupKeyFromGroup(group contracts.DiagnosticGroup) groupKey {
return groupKey{disposition: group.Disposition, category: group.Category, reasonCode: group.ReasonCode, origin: group.Origin}
}

View File

@@ -0,0 +1,89 @@
package diagnostics
import (
"testing"
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
)
func TestAggregatorMergesByOriginAndPreservesFirstOccurrenceOrder(t *testing.T) {
aggregator := Aggregator{}
for _, group := range []contracts.DiagnosticGroup{
groupForAggregation("first", "message one", contracts.DiagnosticDispositionAdvisory, contracts.DiagnosticCategoryDataQuality, "dnd/spells"),
groupForAggregation("second", "message two", contracts.DiagnosticDispositionWarning, contracts.DiagnosticCategoryFallback, "dnd/items"),
groupForAggregation("third", "message three", contracts.DiagnosticDispositionAdvisory, contracts.DiagnosticCategoryDataQuality, "dnd/spells"),
} {
if err := aggregator.Add(group); err != nil {
t.Fatalf("Add() error = %v", err)
}
}
collection := aggregator.Collection()
if len(collection.Groups) != 2 {
t.Fatalf("group count = %d, want 2", len(collection.Groups))
}
if got := []string{collection.Groups[0].Origin.ModuleKey, collection.Groups[1].Origin.ModuleKey}; !equalStrings(got, []string{"dnd/spells", "dnd/items"}) {
t.Fatalf("group order = %#v, want first occurrence order", got)
}
if collection.Groups[0].OccurrenceCount != 2 || collection.Groups[0].OmittedSampleCount != 0 {
t.Fatalf("merged group = %#v, want two represented occurrences", collection.Groups[0])
}
if got := []string{collection.Groups[0].Samples[0].Scope, collection.Groups[0].Samples[1].Scope}; !equalStrings(got, []string{"first", "third"}) {
t.Fatalf("merged samples = %#v, want first occurrence order", got)
}
}
func TestAggregatorEnforcesWarningBoundAndTruncatesOnlyNonWarnings(t *testing.T) {
warnings := Aggregator{}
for index := 0; index < MaxWarningGroups; index++ {
group := groupForAggregation("scope", "message", contracts.DiagnosticDispositionWarning, contracts.DiagnosticCategoryFallback, "module")
group.ReasonCode = "warning-" + string(rune('a'+index))
if err := warnings.Add(group); err != nil {
t.Fatalf("warning Add(%d) error = %v", index, err)
}
}
if err := warnings.Add(groupForAggregation("overflow", "overflow", contracts.DiagnosticDispositionWarning, contracts.DiagnosticCategoryFallback, "overflow")); err == nil {
t.Fatal("warning overflow error = nil, want error")
}
nonWarnings := Aggregator{}
for index := 0; index < MaxNonWarningGroups+3; index++ {
group := groupForAggregation("scope", "message", contracts.DiagnosticDispositionAdvisory, contracts.DiagnosticCategoryDataQuality, "module")
group.ReasonCode = "advisory-" + string(rune('a'+index))
if err := nonWarnings.Add(group); err != nil {
t.Fatalf("non-warning Add(%d) error = %v", index, err)
}
}
collection := nonWarnings.Collection()
if len(collection.Groups) != MaxNonWarningGroups || !collection.Truncated || collection.UnrepresentedOccurrenceCount != 3 {
t.Fatalf("collection = %#v, want bounded non-warning groups and three unrepresented occurrences", collection)
}
}
func TestAggregatorRejectsOccurrenceTotalOverflow(t *testing.T) {
aggregator := Aggregator{}
first := groupForAggregation("first", "first", contracts.DiagnosticDispositionAdvisory, contracts.DiagnosticCategoryDataQuality, "first")
first.OccurrenceCount = maximumInt()
first.OmittedSampleCount = maximumInt() - 1
if err := aggregator.Add(first); err != nil {
t.Fatal(err)
}
second := groupForAggregation("second", "second", contracts.DiagnosticDispositionAdvisory, contracts.DiagnosticCategoryDataQuality, "second")
if err := aggregator.Add(second); err == nil {
t.Fatal("Add() occurrence total overflow error = nil")
}
collection := aggregator.Collection()
if len(collection.Groups) != 1 || collection.Groups[0].OccurrenceCount != maximumInt() {
t.Fatalf("collection changed after rejected overflow = %#v", collection)
}
}
func groupForAggregation(scope string, message string, disposition contracts.DiagnosticDisposition, category contracts.DiagnosticCategory, module string) contracts.DiagnosticGroup {
return contracts.DiagnosticGroup{
Disposition: disposition,
Category: category,
ReasonCode: "source_unrelated",
Origin: contracts.DiagnosticOrigin{Stage: contracts.DiagnosticOriginStageExtract, StepID: "extract", LaneID: "spells", ModuleKey: module},
OccurrenceCount: 1,
Samples: []contracts.DiagnosticSample{{Scope: scope, Message: message}},
}
}

View File

@@ -0,0 +1,93 @@
// Package diagnostics provides bounded local grouping for producer and
// validator diagnostic results.
package diagnostics
import (
"errors"
"fmt"
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
)
// Collector merges local producer diagnostics by their semantic identity. Its
// zero value is ready for use.
type Collector struct {
diagnostics []contracts.ProducerDiagnostic
indices map[key]int
}
// NewCollector returns an empty local diagnostic collector.
func NewCollector() *Collector {
return &Collector{}
}
// Add validates and merges one producer diagnostic. Every occurrence remains
// counted, while the first three distinct samples in input order are retained.
func (collector *Collector) Add(diagnostic contracts.ProducerDiagnostic) error {
if err := diagnostic.Validate(); err != nil {
return fmt.Errorf("producer diagnostic: %w", err)
}
if collector.indices == nil {
collector.indices = make(map[key]int)
}
diagnosticKey := key{disposition: diagnostic.Disposition, category: diagnostic.Category, reasonCode: diagnostic.ReasonCode}
index, exists := collector.indices[diagnosticKey]
if !exists {
if len(collector.diagnostics) >= contracts.MaxProducerDiagnosticGroups {
return errors.New("producer diagnostics exceed maximum group count")
}
collector.indices[diagnosticKey] = len(collector.diagnostics)
collector.diagnostics = append(collector.diagnostics, contracts.CloneProducerDiagnostics([]contracts.ProducerDiagnostic{diagnostic})[0])
return nil
}
current := &collector.diagnostics[index]
if diagnostic.OccurrenceCount > maximumInt()-current.OccurrenceCount {
return errors.New("producer diagnostic occurrence count overflow")
}
current.OccurrenceCount += diagnostic.OccurrenceCount
for _, sample := range diagnostic.Samples {
if len(current.Samples) == contracts.MaxDiagnosticSamples || containsSample(current.Samples, sample) {
continue
}
current.Samples = append(current.Samples, cloneSample(sample))
}
current.OmittedSampleCount = current.OccurrenceCount - len(current.Samples)
return nil
}
// Diagnostics returns an independently owned snapshot in first-occurrence
// order.
func (collector *Collector) Diagnostics() []contracts.ProducerDiagnostic {
if collector == nil {
return nil
}
return contracts.CloneProducerDiagnostics(collector.diagnostics)
}
func containsSample(samples []contracts.DiagnosticSample, candidate contracts.DiagnosticSample) bool {
for _, sample := range samples {
if sample.Scope == candidate.Scope && sample.Message == candidate.Message {
return true
}
}
return false
}
func cloneSample(sample contracts.DiagnosticSample) contracts.DiagnosticSample {
if sample.ChunkIndex != nil {
chunkIndex := *sample.ChunkIndex
sample.ChunkIndex = &chunkIndex
}
return sample
}
func maximumInt() int {
return int(^uint(0) >> 1)
}
type key struct {
disposition contracts.DiagnosticDisposition
category contracts.DiagnosticCategory
reasonCode string
}

View File

@@ -0,0 +1,103 @@
package diagnostics
import (
"testing"
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
)
func TestCollectorCountsOccurrencesAndRetainsDistinctSamplesInOrder(t *testing.T) {
collector := NewCollector()
for _, diagnostic := range []contracts.ProducerDiagnostic{
advisory("one", "first"),
advisory("one", "first"),
advisory("two", "second"),
advisory("three", "third"),
advisory("four", "fourth"),
} {
if err := collector.Add(diagnostic); err != nil {
t.Fatalf("Add() error = %v", err)
}
}
diagnostics := collector.Diagnostics()
if len(diagnostics) != 1 {
t.Fatalf("group count = %d, want 1", len(diagnostics))
}
group := diagnostics[0]
if group.OccurrenceCount != 5 || group.OmittedSampleCount != 2 {
t.Fatalf("group counts = %#v, want five occurrences and two omitted samples", group)
}
if got := []string{group.Samples[0].Scope, group.Samples[1].Scope, group.Samples[2].Scope}; !equalStrings(got, []string{"one", "two", "three"}) {
t.Fatalf("sample order = %#v, want first three distinct samples", got)
}
}
func TestCollectorSeparatesGroupsAndRejectsInvalidOrExcessiveGroups(t *testing.T) {
collector := NewCollector()
if err := collector.Add(advisory("one", "first")); err != nil {
t.Fatalf("Add() error = %v", err)
}
warning := advisory("two", "second")
warning.Disposition = contracts.DiagnosticDispositionWarning
warning.Category = contracts.DiagnosticCategoryFallback
if err := collector.Add(warning); err != nil {
t.Fatalf("Add() error = %v", err)
}
if got := len(collector.Diagnostics()); got != 2 {
t.Fatalf("group count = %d, want 2", got)
}
invalid := advisory("bad", "bad")
invalid.ReasonCode = ""
if err := collector.Add(invalid); err == nil {
t.Fatal("Add() error = nil, want invalid diagnostic error")
}
limited := NewCollector()
for index := 0; index < contracts.MaxProducerDiagnosticGroups; index++ {
diagnostic := advisory("scope", "message")
diagnostic.ReasonCode = "reason-" + string(rune('a'+index))
if err := limited.Add(diagnostic); err != nil {
t.Fatalf("Add(%d) error = %v", index, err)
}
}
if err := limited.Add(advisory("overflow", "overflow")); err == nil {
t.Fatal("Add() error = nil, want local group limit error")
}
}
func TestCollectorReturnsIndependentSnapshots(t *testing.T) {
collector := NewCollector()
if err := collector.Add(advisory("scope", "message")); err != nil {
t.Fatalf("Add() error = %v", err)
}
first := collector.Diagnostics()
first[0].Samples[0].Message = "changed"
second := collector.Diagnostics()
if second[0].Samples[0].Message != "message" {
t.Fatalf("collector snapshot changed = %#v", second)
}
}
func advisory(scope string, message string) contracts.ProducerDiagnostic {
return contracts.ProducerDiagnostic{
Disposition: contracts.DiagnosticDispositionAdvisory,
Category: contracts.DiagnosticCategoryDataQuality,
ReasonCode: "source_unrelated",
OccurrenceCount: 1,
Samples: []contracts.DiagnosticSample{{Scope: scope, Message: message}},
}
}
func equalStrings(left []string, right []string) bool {
if len(left) != len(right) {
return false
}
for index := range left {
if left[index] != right[index] {
return false
}
}
return true
}

View File

@@ -28,14 +28,14 @@ type CheckpointRecorder interface {
SourceSucceeded(moduleKey string, doc *source.SourceDocument) error SourceSucceeded(moduleKey string, doc *source.SourceDocument) error
SourceFailed(moduleKey string, err error) error SourceFailed(moduleKey string, err error) error
ExtractRunning(laneID string, moduleKey string, dependencies []CheckpointFingerprint) error ExtractRunning(laneID string, moduleKey string, dependencies []CheckpointFingerprint) error
ExtractSucceeded(laneID string, moduleKey string, dependencies []CheckpointFingerprint, outputs []CheckpointArtifact, rejected []contracts.RejectedOutput, warnings []contracts.Warning) error ExtractSucceeded(laneID string, moduleKey string, dependencies []CheckpointFingerprint, outputs []CheckpointArtifact, rejected []contracts.RejectedOutput) error
ExtractFailed(laneID string, moduleKey string, dependencies []CheckpointFingerprint, err error) error ExtractFailed(laneID string, moduleKey string, dependencies []CheckpointFingerprint, err error) error
MergeRunning(laneID string, moduleKey string, dependencies []CheckpointFingerprint) error MergeRunning(laneID string, moduleKey string, dependencies []CheckpointFingerprint) error
MergeSucceeded(laneID string, moduleKey string, dependencies []CheckpointFingerprint, output CheckpointArtifact, warnings []contracts.Warning) error MergeSucceeded(laneID string, moduleKey string, dependencies []CheckpointFingerprint, output CheckpointArtifact) error
MergeRejected(laneID string, moduleKey string, dependencies []CheckpointFingerprint, rejected contracts.RejectedOutput) error MergeRejected(laneID string, moduleKey string, dependencies []CheckpointFingerprint, rejected contracts.RejectedOutput) error
MergeFailed(laneID string, moduleKey string, dependencies []CheckpointFingerprint, err error) error MergeFailed(laneID string, moduleKey string, dependencies []CheckpointFingerprint, err error) error
NormalizeRunning(laneID string, moduleKey string, dependencies []CheckpointFingerprint) error NormalizeRunning(laneID string, moduleKey string, dependencies []CheckpointFingerprint) error
NormalizeSucceeded(laneID string, moduleKey string, dependencies []CheckpointFingerprint, output CheckpointArtifact, warnings []contracts.Warning) error NormalizeSucceeded(laneID string, moduleKey string, dependencies []CheckpointFingerprint, output CheckpointArtifact) error
NormalizeRejected(laneID string, moduleKey string, dependencies []CheckpointFingerprint, rejected contracts.RejectedOutput) error NormalizeRejected(laneID string, moduleKey string, dependencies []CheckpointFingerprint, rejected contracts.RejectedOutput) error
NormalizeFailed(laneID string, moduleKey string, dependencies []CheckpointFingerprint, err error) error NormalizeFailed(laneID string, moduleKey string, dependencies []CheckpointFingerprint, err error) error
} }
@@ -45,14 +45,14 @@ type CheckpointRecorder interface {
// remain available for callers that do not have step context. // remain available for callers that do not have step context.
type StepCheckpointRecorder interface { type StepCheckpointRecorder interface {
ExtractRunningForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint) error ExtractRunningForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint) error
ExtractSucceededForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint, outputs []CheckpointArtifact, rejected []contracts.RejectedOutput, warnings []contracts.Warning) error ExtractSucceededForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint, outputs []CheckpointArtifact, rejected []contracts.RejectedOutput) error
ExtractFailedForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint, err error) error ExtractFailedForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint, err error) error
MergeRunningForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint) error MergeRunningForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint) error
MergeSucceededForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint, output CheckpointArtifact, warnings []contracts.Warning) error MergeSucceededForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint, output CheckpointArtifact) error
MergeRejectedForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint, rejected contracts.RejectedOutput) error MergeRejectedForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint, rejected contracts.RejectedOutput) error
MergeFailedForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint, err error) error MergeFailedForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint, err error) error
NormalizeRunningForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint) error NormalizeRunningForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint) error
NormalizeSucceededForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint, output CheckpointArtifact, warnings []contracts.Warning) error NormalizeSucceededForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint, output CheckpointArtifact) error
NormalizeRejectedForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint, rejected contracts.RejectedOutput) error NormalizeRejectedForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint, rejected contracts.RejectedOutput) error
NormalizeFailedForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint, err error) error NormalizeFailedForStep(stepID, laneID string, moduleKey string, dependencies []CheckpointFingerprint, err error) error
} }
@@ -279,21 +279,27 @@ type CheckpointArtifact struct {
ChunkRef source.SourceRef ChunkRef source.SourceRef
Artifact contracts.SerializedArtifact Artifact contracts.SerializedArtifact
SchemaDigest string SchemaDigest string
Diagnostics []CheckpointDiagnostic
}
// CheckpointDiagnostic stores a producer-local diagnostic alongside a
// reusable artifact. The runner supplies the current run's origin when it
// promotes this value into a diagnostic group.
type CheckpointDiagnostic struct {
Diagnostic contracts.ProducerDiagnostic `json:"diagnostic"`
ValidatorKey string `json:"validator_key,omitempty"`
} }
type ExtractCheckpoint struct { type ExtractCheckpoint struct {
Outputs []CheckpointArtifact Outputs []CheckpointArtifact
Rejected []contracts.RejectedOutput Rejected []contracts.RejectedOutput
Warnings []contracts.Warning
} }
type MergeCheckpoint struct { type MergeCheckpoint struct {
Output CheckpointArtifact Output CheckpointArtifact
Warnings []contracts.Warning
} }
type NormalizeCheckpoint struct { type NormalizeCheckpoint struct {
Output CheckpointArtifact Output CheckpointArtifact
Warnings []contracts.Warning
} }
type CheckpointLoader interface { type CheckpointLoader interface {
@@ -326,14 +332,14 @@ func (noopCheckpointRecorder) SourceFailed(string, error) error
func (noopCheckpointRecorder) ExtractRunning(string, string, []CheckpointFingerprint) error { func (noopCheckpointRecorder) ExtractRunning(string, string, []CheckpointFingerprint) error {
return nil return nil
} }
func (noopCheckpointRecorder) ExtractSucceeded(string, string, []CheckpointFingerprint, []CheckpointArtifact, []contracts.RejectedOutput, []contracts.Warning) error { func (noopCheckpointRecorder) ExtractSucceeded(string, string, []CheckpointFingerprint, []CheckpointArtifact, []contracts.RejectedOutput) error {
return nil return nil
} }
func (noopCheckpointRecorder) ExtractFailed(string, string, []CheckpointFingerprint, error) error { func (noopCheckpointRecorder) ExtractFailed(string, string, []CheckpointFingerprint, error) error {
return nil return nil
} }
func (noopCheckpointRecorder) MergeRunning(string, string, []CheckpointFingerprint) error { return nil } func (noopCheckpointRecorder) MergeRunning(string, string, []CheckpointFingerprint) error { return nil }
func (noopCheckpointRecorder) MergeSucceeded(string, string, []CheckpointFingerprint, CheckpointArtifact, []contracts.Warning) error { func (noopCheckpointRecorder) MergeSucceeded(string, string, []CheckpointFingerprint, CheckpointArtifact) error {
return nil return nil
} }
func (noopCheckpointRecorder) MergeRejected(string, string, []CheckpointFingerprint, contracts.RejectedOutput) error { func (noopCheckpointRecorder) MergeRejected(string, string, []CheckpointFingerprint, contracts.RejectedOutput) error {
@@ -345,7 +351,7 @@ func (noopCheckpointRecorder) MergeFailed(string, string, []CheckpointFingerprin
func (noopCheckpointRecorder) NormalizeRunning(string, string, []CheckpointFingerprint) error { func (noopCheckpointRecorder) NormalizeRunning(string, string, []CheckpointFingerprint) error {
return nil return nil
} }
func (noopCheckpointRecorder) NormalizeSucceeded(string, string, []CheckpointFingerprint, CheckpointArtifact, []contracts.Warning) error { func (noopCheckpointRecorder) NormalizeSucceeded(string, string, []CheckpointFingerprint, CheckpointArtifact) error {
return nil return nil
} }
func (noopCheckpointRecorder) NormalizeRejected(string, string, []CheckpointFingerprint, contracts.RejectedOutput) error { func (noopCheckpointRecorder) NormalizeRejected(string, string, []CheckpointFingerprint, contracts.RejectedOutput) error {
@@ -378,11 +384,11 @@ func checkpointExtractRunning(recorder CheckpointRecorder, stepID, laneID, modul
} }
return recorder.ExtractRunning(laneID, moduleKey, deps) return recorder.ExtractRunning(laneID, moduleKey, deps)
} }
func checkpointExtractSucceeded(recorder CheckpointRecorder, stepID, laneID, moduleKey string, deps []CheckpointFingerprint, outputs []CheckpointArtifact, rejected []contracts.RejectedOutput, warnings []contracts.Warning) error { func checkpointExtractSucceeded(recorder CheckpointRecorder, stepID, laneID, moduleKey string, deps []CheckpointFingerprint, outputs []CheckpointArtifact, rejected []contracts.RejectedOutput) error {
if stepAware, ok := recorder.(StepCheckpointRecorder); ok { if stepAware, ok := recorder.(StepCheckpointRecorder); ok {
return stepAware.ExtractSucceededForStep(stepID, laneID, moduleKey, deps, outputs, rejected, warnings) return stepAware.ExtractSucceededForStep(stepID, laneID, moduleKey, deps, outputs, rejected)
} }
return recorder.ExtractSucceeded(laneID, moduleKey, deps, outputs, rejected, warnings) return recorder.ExtractSucceeded(laneID, moduleKey, deps, outputs, rejected)
} }
func checkpointExtractFailed(recorder CheckpointRecorder, stepID, laneID, moduleKey string, deps []CheckpointFingerprint, err error) error { func checkpointExtractFailed(recorder CheckpointRecorder, stepID, laneID, moduleKey string, deps []CheckpointFingerprint, err error) error {
if stepAware, ok := recorder.(StepCheckpointRecorder); ok { if stepAware, ok := recorder.(StepCheckpointRecorder); ok {
@@ -396,11 +402,11 @@ func checkpointMergeRunning(recorder CheckpointRecorder, stepID, laneID, moduleK
} }
return recorder.MergeRunning(laneID, moduleKey, deps) return recorder.MergeRunning(laneID, moduleKey, deps)
} }
func checkpointMergeSucceeded(recorder CheckpointRecorder, stepID, laneID, moduleKey string, deps []CheckpointFingerprint, output CheckpointArtifact, warnings []contracts.Warning) error { func checkpointMergeSucceeded(recorder CheckpointRecorder, stepID, laneID, moduleKey string, deps []CheckpointFingerprint, output CheckpointArtifact) error {
if stepAware, ok := recorder.(StepCheckpointRecorder); ok { if stepAware, ok := recorder.(StepCheckpointRecorder); ok {
return stepAware.MergeSucceededForStep(stepID, laneID, moduleKey, deps, output, warnings) return stepAware.MergeSucceededForStep(stepID, laneID, moduleKey, deps, output)
} }
return recorder.MergeSucceeded(laneID, moduleKey, deps, output, warnings) return recorder.MergeSucceeded(laneID, moduleKey, deps, output)
} }
func checkpointMergeRejected(recorder CheckpointRecorder, stepID, laneID, moduleKey string, deps []CheckpointFingerprint, rejected contracts.RejectedOutput) error { func checkpointMergeRejected(recorder CheckpointRecorder, stepID, laneID, moduleKey string, deps []CheckpointFingerprint, rejected contracts.RejectedOutput) error {
if stepAware, ok := recorder.(StepCheckpointRecorder); ok { if stepAware, ok := recorder.(StepCheckpointRecorder); ok {
@@ -420,11 +426,11 @@ func checkpointNormalizeRunning(recorder CheckpointRecorder, stepID, laneID, mod
} }
return recorder.NormalizeRunning(laneID, moduleKey, deps) return recorder.NormalizeRunning(laneID, moduleKey, deps)
} }
func checkpointNormalizeSucceeded(recorder CheckpointRecorder, stepID, laneID, moduleKey string, deps []CheckpointFingerprint, output CheckpointArtifact, warnings []contracts.Warning) error { func checkpointNormalizeSucceeded(recorder CheckpointRecorder, stepID, laneID, moduleKey string, deps []CheckpointFingerprint, output CheckpointArtifact) error {
if stepAware, ok := recorder.(StepCheckpointRecorder); ok { if stepAware, ok := recorder.(StepCheckpointRecorder); ok {
return stepAware.NormalizeSucceededForStep(stepID, laneID, moduleKey, deps, output, warnings) return stepAware.NormalizeSucceededForStep(stepID, laneID, moduleKey, deps, output)
} }
return recorder.NormalizeSucceeded(laneID, moduleKey, deps, output, warnings) return recorder.NormalizeSucceeded(laneID, moduleKey, deps, output)
} }
func checkpointNormalizeRejected(recorder CheckpointRecorder, stepID, laneID, moduleKey string, deps []CheckpointFingerprint, rejected contracts.RejectedOutput) error { func checkpointNormalizeRejected(recorder CheckpointRecorder, stepID, laneID, moduleKey string, deps []CheckpointFingerprint, rejected contracts.RejectedOutput) error {
if stepAware, ok := recorder.(StepCheckpointRecorder); ok { if stepAware, ok := recorder.(StepCheckpointRecorder); ok {

View File

@@ -8,7 +8,7 @@ import (
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts" "gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
) )
const ChunkPlanSchemaVersion = "notarius.chunk-plan.v2" const ChunkPlanSchemaVersion = "notarius.chunk-plan.v3"
type ChunkPlanProducer struct { type ChunkPlanProducer struct {
InputModule string `json:"input_module"` InputModule string `json:"input_module"`
@@ -24,7 +24,7 @@ type ChunkPlanRecord struct {
PlanDigest string `json:"plan_digest"` PlanDigest string `json:"plan_digest"`
Plan source.ChunkPlan `json:"plan"` Plan source.ChunkPlan `json:"plan"`
Producer ChunkPlanProducer `json:"producer"` Producer ChunkPlanProducer `json:"producer"`
Warnings []contracts.Warning `json:"warnings,omitempty"` Diagnostics []contracts.ProducerDiagnostic `json:"diagnostics,omitempty"`
CreatedAt time.Time `json:"created_at"` CreatedAt time.Time `json:"created_at"`
} }

View File

@@ -56,7 +56,6 @@ type debugBinaryEnvelope struct {
ContentDigest string `json:"content_digest,omitempty"` ContentDigest string `json:"content_digest,omitempty"`
MediaType string `json:"media_type,omitempty"` MediaType string `json:"media_type,omitempty"`
Metadata map[string]any `json:"metadata,omitempty"` Metadata map[string]any `json:"metadata,omitempty"`
Warnings []contracts.Warning `json:"warnings,omitempty"`
} }
type debugSourceInput struct { type debugSourceInput struct {
@@ -483,14 +482,13 @@ func writeProducerTerminalDebug(recorder DebugRecorder, name string, terminal pr
}) })
} }
func debugContentEnvelope(content []byte, mediaType string, metadata map[string]any, warnings []contracts.Warning) debugBinaryEnvelope { func debugContentEnvelope(content []byte, mediaType string, metadata map[string]any, _ any) debugBinaryEnvelope {
content = redactSecretBytes(content) content = redactSecretBytes(content)
return debugBinaryEnvelope{ return debugBinaryEnvelope{
ContentBase64: base64.StdEncoding.EncodeToString(content), ContentBase64: base64.StdEncoding.EncodeToString(content),
ContentDigest: debugContentDigest(content), ContentDigest: debugContentDigest(content),
MediaType: mediaType, MediaType: mediaType,
Metadata: redactSensitiveMap(metadata), Metadata: redactSensitiveMap(metadata),
Warnings: cloneWarnings(warnings),
} }
} }
@@ -731,20 +729,9 @@ func debugValidationResultEnvelope(result contracts.ValidationResult) contracts.
result.Message = string(redactSecretBytes([]byte(result.Message))) result.Message = string(redactSecretBytes([]byte(result.Message)))
result.CorrectionGuidance = "" result.CorrectionGuidance = ""
result.DiagnosticArtifactPath = string(redactSecretBytes([]byte(result.DiagnosticArtifactPath))) result.DiagnosticArtifactPath = string(redactSecretBytes([]byte(result.DiagnosticArtifactPath)))
for i := range result.Warnings {
result.Warnings[i].Message = string(redactSecretBytes([]byte(result.Warnings[i].Message)))
}
return result return result
} }
func debugWarningEnvelopes(warnings []contracts.Warning) []contracts.Warning {
out := cloneWarnings(warnings)
for i := range out {
out[i].Message = string(redactSecretBytes([]byte(out[i].Message)))
}
return out
}
func debugRejectedOutputEnvelope(rejected contracts.RejectedOutput) contracts.RejectedOutput { func debugRejectedOutputEnvelope(rejected contracts.RejectedOutput) contracts.RejectedOutput {
rejected.Message = string(redactSecretBytes([]byte(rejected.Message))) rejected.Message = string(redactSecretBytes([]byte(rejected.Message)))
rejected.DiagnosticArtifactPath = string(redactSecretBytes([]byte(rejected.DiagnosticArtifactPath))) rejected.DiagnosticArtifactPath = string(redactSecretBytes([]byte(rejected.DiagnosticArtifactPath)))

View File

@@ -0,0 +1,124 @@
package pipeline
import (
"fmt"
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
frameworkdiagnostics "gitea.maximumdirect.net/eric/notarius/internal/framework/diagnostics"
)
func terminalDiagnosticGroups(terminal producerAttemptTerminal, origin contracts.DiagnosticOrigin, chunk *source.Chunk) ([]contracts.DiagnosticGroup, error) {
return promoteCheckpointDiagnostics(terminalCheckpointDiagnostics(terminal), origin, chunk)
}
func terminalCheckpointDiagnostics(terminal producerAttemptTerminal) []CheckpointDiagnostic {
diagnostics := make([]CheckpointDiagnostic, 0, len(terminal.Diagnostics))
for _, diagnostic := range terminal.Diagnostics {
diagnostics = append(diagnostics, CheckpointDiagnostic{Diagnostic: diagnostic})
}
for _, record := range terminal.Validation.Diagnostics() {
diagnostics = append(diagnostics, CheckpointDiagnostic{Diagnostic: record.diagnostic, ValidatorKey: record.validatorName})
}
if terminal.Action == producerTerminalIncompleteAccepted {
for _, record := range incompleteValidationDiagnostics(terminal.Validation) {
diagnostics = append(diagnostics, CheckpointDiagnostic{Diagnostic: record.diagnostic, ValidatorKey: record.validatorName})
}
}
return cloneCheckpointDiagnostics(diagnostics)
}
func promoteCheckpointDiagnostics(diagnostics []CheckpointDiagnostic, origin contracts.DiagnosticOrigin, chunk *source.Chunk) ([]contracts.DiagnosticGroup, error) {
groups := make([]contracts.DiagnosticGroup, 0, len(diagnostics))
for _, record := range diagnostics {
validatorOrigin := origin
validatorOrigin.ValidatorKey = record.ValidatorKey
promoted, err := promoteProducerDiagnostics([]contracts.ProducerDiagnostic{record.Diagnostic}, validatorOrigin, chunk)
if err != nil {
return nil, fmt.Errorf("checkpoint diagnostic: %w", err)
}
groups = append(groups, promoted...)
}
return groups, nil
}
func cloneCheckpointDiagnostics(diagnostics []CheckpointDiagnostic) []CheckpointDiagnostic {
if len(diagnostics) == 0 {
return nil
}
cloned := make([]CheckpointDiagnostic, len(diagnostics))
for index, diagnostic := range diagnostics {
cloned[index] = CheckpointDiagnostic{
Diagnostic: contracts.CloneProducerDiagnostics([]contracts.ProducerDiagnostic{diagnostic.Diagnostic})[0],
ValidatorKey: diagnostic.ValidatorKey,
}
}
return cloned
}
func promoteProducerDiagnostics(diagnostics []contracts.ProducerDiagnostic, origin contracts.DiagnosticOrigin, chunk *source.Chunk) ([]contracts.DiagnosticGroup, error) {
if err := contracts.ValidateProducerDiagnostics(diagnostics); err != nil {
return nil, err
}
if len(diagnostics) == 0 {
return nil, nil
}
groups := make([]contracts.DiagnosticGroup, len(diagnostics))
for index, diagnostic := range diagnostics {
group := contracts.DiagnosticGroup{
Disposition: diagnostic.Disposition,
Category: diagnostic.Category,
ReasonCode: diagnostic.ReasonCode,
Origin: origin,
OccurrenceCount: diagnostic.OccurrenceCount,
Samples: cloneDiagnosticSamples(diagnostic.Samples, chunk),
OmittedSampleCount: diagnostic.OmittedSampleCount,
}
if err := group.Validate(); err != nil {
return nil, err
}
groups[index] = group
}
return groups, nil
}
func cloneDiagnosticSamples(samples []contracts.DiagnosticSample, chunk *source.Chunk) []contracts.DiagnosticSample {
if len(samples) == 0 {
return nil
}
cloned := make([]contracts.DiagnosticSample, len(samples))
for index, sample := range samples {
if chunk != nil {
chunkIndex := chunk.Index
sample.ChunkID = chunk.ID
sample.ChunkIndex = &chunkIndex
} else if sample.ChunkIndex != nil {
chunkIndex := *sample.ChunkIndex
sample.ChunkIndex = &chunkIndex
}
cloned[index] = sample
}
return cloned
}
func appendDiagnosticGroups(output *RunOutput, groups []contracts.DiagnosticGroup) {
if output == nil || len(groups) == 0 {
return
}
cloned := contracts.CloneDiagnosticCollection(contracts.DiagnosticCollection{Groups: groups}).Groups
output.diagnosticGroups = append(output.diagnosticGroups, cloned...)
}
func finalizeDiagnostics(output *RunOutput) error {
if output == nil {
return nil
}
aggregator := frameworkdiagnostics.Aggregator{}
for _, group := range output.diagnosticGroups {
if err := aggregator.Add(group); err != nil {
return err
}
}
output.Diagnostics = aggregator.Collection()
return nil
}

View File

@@ -0,0 +1,75 @@
package pipeline
import (
"context"
"testing"
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
)
func TestTerminalDiagnosticGroupsDiscardSupersededAttemptDiagnostics(t *testing.T) {
terminal, err := runProducerAttempts(context.Background(), producerAttemptConfig{Retries: 1, Policy: DefaultValidationPolicy()}, func(_ context.Context, request producerAttemptRequest) (producerAttemptOutput, error) {
if request.Number == 1 {
return producerAttemptOutput{Value: "first", Diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic("discarded", "discarded")}, Candidate: attemptCandidate(t, "first")}, nil
}
return producerAttemptOutput{Value: "second", Diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic("terminal", "terminal")}}, nil
}, func(_ context.Context, output producerAttemptOutput) (validationReport, error) {
if output.Value == "first" {
return validationReport{records: []validationRecord{{validatorName: "validator", outcome: validationRejected, reasonCode: "invalid", correctionGuidance: "Correct it."}}}, nil
}
return validationReport{}, nil
})
if err != nil {
t.Fatalf("runProducerAttempts() error = %v", err)
}
groups, err := terminalDiagnosticGroups(terminal, contracts.DiagnosticOrigin{Stage: contracts.DiagnosticOriginStageMerge, StepID: "step", LaneID: "lane", ModuleKey: "module"}, nil)
if err != nil {
t.Fatalf("terminalDiagnosticGroups() error = %v", err)
}
if len(groups) != 1 || groups[0].Samples[0].Scope != "terminal" {
t.Fatalf("groups = %#v, want only terminal attempt diagnostics", groups)
}
}
func TestTerminalDiagnosticGroupsPreserveValidatorOrigin(t *testing.T) {
terminal := producerAttemptTerminal{Validation: validationReport{records: []validationRecord{{
validatorName: "dnd/spells/source-relatedness",
outcome: validationApproved,
diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic("spells[0]", "not found")},
}}}}
groups, err := terminalDiagnosticGroups(terminal, contracts.DiagnosticOrigin{Stage: contracts.DiagnosticOriginStageExtract, StepID: "extract", LaneID: "spells", ModuleKey: "dnd/spells"}, nil)
if err != nil {
t.Fatalf("terminalDiagnosticGroups() error = %v", err)
}
if len(groups) != 1 || groups[0].Origin.ValidatorKey != "dnd/spells/source-relatedness" {
t.Fatalf("groups = %#v, want validator origin", groups)
}
}
func TestAppendDiagnosticGroupsRetainsOnlyStructuredDiagnostics(t *testing.T) {
output := RunOutput{}
appendDiagnosticGroups(&output, []contracts.DiagnosticGroup{{
Disposition: contracts.DiagnosticDispositionAdvisory,
Category: contracts.DiagnosticCategoryDataQuality,
ReasonCode: "source_unrelated",
Origin: contracts.DiagnosticOrigin{Stage: contracts.DiagnosticOriginStageExtract, StepID: "extract", LaneID: "spells", ModuleKey: "dnd/spells"},
OccurrenceCount: 2,
Samples: []contracts.DiagnosticSample{{Scope: "spells[0]", Message: "not found"}, {Scope: "spells[1]", Message: "also not found"}},
}})
if err := finalizeDiagnostics(&output); err != nil {
t.Fatalf("finalizeDiagnostics() error = %v", err)
}
if len(output.Diagnostics.Groups) != 1 || output.Diagnostics.Groups[0].OccurrenceCount != 2 {
t.Fatalf("diagnostics = %#v, want grouped collection", output.Diagnostics)
}
}
func producerDiagnostic(scope string, message string) contracts.ProducerDiagnostic {
return contracts.ProducerDiagnostic{
Disposition: contracts.DiagnosticDispositionAdvisory,
Category: contracts.DiagnosticCategoryDataQuality,
ReasonCode: "source_unrelated",
OccurrenceCount: 1,
Samples: []contracts.DiagnosticSample{{Scope: scope, Message: message}},
}
}

View File

@@ -71,7 +71,7 @@ func RegisterExtractorBuilder[T any](registry *ExtractorRegistry, spec ModuleSpe
if err != nil { if err != nil {
return erasedTypedResult{}, fmt.Errorf("clone extraction model candidate: %w", err) return erasedTypedResult{}, fmt.Errorf("clone extraction model candidate: %w", err)
} }
return erasedTypedResult{Value: result.Value, Warnings: cloneWarnings(result.Warnings), ModelCandidate: candidate}, nil return erasedTypedResult{Value: result.Value, Diagnostics: contracts.CloneProducerDiagnostics(result.Diagnostics), ModelCandidate: candidate}, nil
}} }}
if registry.typedEntries == nil { if registry.typedEntries == nil {
registry.typedEntries = map[string]typedExtractorEntry{} registry.typedEntries = map[string]typedExtractorEntry{}

View File

@@ -97,7 +97,7 @@ func RegisterMergerBuilder[T any](registry *MergerRegistry, spec ModuleSpec, val
if err != nil { if err != nil {
return erasedTypedResult{}, fmt.Errorf("clone merge model candidate: %w", err) return erasedTypedResult{}, fmt.Errorf("clone merge model candidate: %w", err)
} }
return erasedTypedResult{Value: result.Value, Warnings: cloneWarnings(result.Warnings), ModelCandidate: candidate}, nil return erasedTypedResult{Value: result.Value, Diagnostics: contracts.CloneProducerDiagnostics(result.Diagnostics), ModelCandidate: candidate}, nil
}, },
} }
return nil return nil

View File

@@ -30,5 +30,14 @@ func validateNormalizeRetry(retry *contracts.NormalizeRetry) error {
if len(retry.Message) > contracts.MaxNormalizeRetryMessageBytes { if len(retry.Message) > contracts.MaxNormalizeRetryMessageBytes {
return errors.New("normalize retry directive message exceeds maximum length") return errors.New("normalize retry directive message exceeds maximum length")
} }
if !utf8.ValidString(retry.CorrectionGuidance) {
return errors.New("normalize retry directive correction guidance has invalid UTF-8")
}
if retry.CorrectionGuidance != "" && strings.TrimSpace(retry.CorrectionGuidance) == "" {
return errors.New("normalize retry directive correction guidance is blank")
}
if len(retry.CorrectionGuidance) > contracts.MaxNormalizeRetryCorrectionGuidanceBytes {
return errors.New("normalize retry directive correction guidance exceeds maximum length")
}
return nil return nil
} }

View File

@@ -88,7 +88,7 @@ func RegisterNormalizerBuilder[T any](registry *NormalizerRegistry, spec ModuleS
if err != nil { if err != nil {
return erasedTypedResult{}, fmt.Errorf("clone normalize model candidate: %w", err) return erasedTypedResult{}, fmt.Errorf("clone normalize model candidate: %w", err)
} }
return erasedTypedResult{Value: result.Value, Warnings: cloneWarnings(result.Warnings), Retry: cloneNormalizeRetry(result.Retry), ModelCandidate: candidate}, nil return erasedTypedResult{Value: result.Value, Diagnostics: contracts.CloneProducerDiagnostics(result.Diagnostics), Retry: cloneNormalizeRetry(result.Retry), ModelCandidate: candidate}, nil
}, },
} }
return nil return nil
@@ -101,7 +101,8 @@ func cloneNormalizeRetry(retry *contracts.NormalizeRetry) *contracts.NormalizeRe
return &contracts.NormalizeRetry{ return &contracts.NormalizeRetry{
ReasonCode: retry.ReasonCode, ReasonCode: retry.ReasonCode,
Message: retry.Message, Message: retry.Message,
FallbackWarnings: cloneWarnings(retry.FallbackWarnings), CorrectionGuidance: retry.CorrectionGuidance,
FallbackDiagnostics: contracts.CloneProducerDiagnostics(retry.FallbackDiagnostics),
} }
} }

View File

@@ -8,26 +8,24 @@ import (
) )
type retryingNotesNormalizer struct { type retryingNotesNormalizer struct {
warnings []contracts.Warning
retry *contracts.NormalizeRetry retry *contracts.NormalizeRetry
} }
func (retryingNotesNormalizer) Key() string { return "test/retry-normalize" } func (retryingNotesNormalizer) Key() string { return "test/retry-normalize" }
func (retryingNotesNormalizer) ReferenceSlots() []contracts.ReferenceSlot { return nil } func (retryingNotesNormalizer) ReferenceSlots() []contracts.ReferenceSlot { return nil }
func (n retryingNotesNormalizer) Normalize(_ context.Context, req contracts.TypedNormalizeRequest[codecNotes]) (contracts.TypedNormalizeResult[codecNotes], error) { func (n retryingNotesNormalizer) Normalize(_ context.Context, req contracts.TypedNormalizeRequest[codecNotes]) (contracts.TypedNormalizeResult[codecNotes], error) {
return contracts.TypedNormalizeResult[codecNotes]{Value: req.MergeOutput.Value, Warnings: n.warnings, Retry: n.retry}, nil return contracts.TypedNormalizeResult[codecNotes]{Value: req.MergeOutput.Value, Retry: n.retry}, nil
} }
func TestNormalizerRegistryErasureClonesRetryDirective(t *testing.T) { func TestNormalizerRegistryErasureClonesRetryDirective(t *testing.T) {
warnings := []contracts.Warning{{Scope: "attempt", ReasonCode: "ordinary", Message: "ordinary warning"}}
retry := &contracts.NormalizeRetry{ retry := &contracts.NormalizeRetry{
ReasonCode: "retryable", ReasonCode: "retryable",
Message: "safe fallback available", Message: "safe fallback available",
FallbackWarnings: []contracts.Warning{{Scope: "fallback", ReasonCode: "omitted", Message: "fallback warning"}}, CorrectionGuidance: "return a complete corrected proposal",
} }
registry := NewNormalizerRegistry() registry := NewNormalizerRegistry()
if err := RegisterNormalizer(registry, ModuleSpec{Key: "test/retry-normalize", Stage: StageNormalize, ExecutionClass: contracts.ExecutionClassDeterministic, ArtifactKind: "test/notes"}, func() (contracts.Normalizer[codecNotes], error) { if err := RegisterNormalizer(registry, ModuleSpec{Key: "test/retry-normalize", Stage: StageNormalize, ExecutionClass: contracts.ExecutionClassDeterministic, ArtifactKind: "test/notes"}, func() (contracts.Normalizer[codecNotes], error) {
return retryingNotesNormalizer{warnings: warnings, retry: retry}, nil return retryingNotesNormalizer{retry: retry}, nil
}); err != nil { }); err != nil {
t.Fatalf("RegisterNormalizer() error = %v", err) t.Fatalf("RegisterNormalizer() error = %v", err)
} }
@@ -43,10 +41,9 @@ func TestNormalizerRegistryErasureClonesRetryDirective(t *testing.T) {
if err != nil { if err != nil {
t.Fatalf("normalize() error = %v", err) t.Fatalf("normalize() error = %v", err)
} }
warnings[0].Message = "mutated"
retry.Message = "mutated" retry.Message = "mutated"
retry.FallbackWarnings[0].Message = "mutated" retry.CorrectionGuidance = "mutated guidance"
if result.Retry == nil || result.Warnings[0].Message != "ordinary warning" || result.Retry.Message != "safe fallback available" || result.Retry.FallbackWarnings[0].Message != "fallback warning" { if result.Retry == nil || result.Retry.Message != "safe fallback available" || result.Retry.CorrectionGuidance != "return a complete corrected proposal" {
t.Fatalf("erased retry result = %#v, want independent warning data", result) t.Fatalf("erased retry result = %#v, want independent retry data", result)
} }
} }

View File

@@ -52,14 +52,18 @@ type producerAttemptRequest struct {
// the current value as a safe fallback if its shared budget is exhausted. // the current value as a safe fallback if its shared budget is exhausted.
// Artifact-specific adapters are responsible for validating and populating it. // Artifact-specific adapters are responsible for validating and populating it.
type producerRetryDirective struct { type producerRetryDirective struct {
FallbackWarnings []contracts.Warning CorrectionGuidance string
FallbackDiagnostics []contracts.ProducerDiagnostic
} }
func (directive *producerRetryDirective) clone() *producerRetryDirective { func (directive *producerRetryDirective) clone() *producerRetryDirective {
if directive == nil { if directive == nil {
return nil return nil
} }
return &producerRetryDirective{FallbackWarnings: cloneWarnings(directive.FallbackWarnings)} return &producerRetryDirective{
CorrectionGuidance: directive.CorrectionGuidance,
FallbackDiagnostics: contracts.CloneProducerDiagnostics(directive.FallbackDiagnostics),
}
} }
// producerAttemptOutput is intentionally artifact-neutral. Value remains // producerAttemptOutput is intentionally artifact-neutral. Value remains
@@ -68,7 +72,7 @@ func (directive *producerRetryDirective) clone() *producerRetryDirective {
type producerAttemptOutput struct { type producerAttemptOutput struct {
Value any Value any
Candidate *contracts.ModelCandidate Candidate *contracts.ModelCandidate
Warnings []contracts.Warning Diagnostics []contracts.ProducerDiagnostic
Retry *producerRetryDirective Retry *producerRetryDirective
} }
@@ -80,7 +84,7 @@ func (output producerAttemptOutput) clone() (producerAttemptOutput, error) {
return producerAttemptOutput{ return producerAttemptOutput{
Value: output.Value, Value: output.Value,
Candidate: candidate, Candidate: candidate,
Warnings: cloneWarnings(output.Warnings), Diagnostics: contracts.CloneProducerDiagnostics(output.Diagnostics),
Retry: output.Retry.clone(), Retry: output.Retry.clone(),
}, nil }, nil
} }
@@ -113,7 +117,7 @@ func (provenance producerAttemptProvenance) clone() producerAttemptProvenance {
type producerAttemptTerminal struct { type producerAttemptTerminal struct {
Action producerTerminalAction Action producerTerminalAction
Value any Value any
Warnings []contracts.Warning Diagnostics []contracts.ProducerDiagnostic
Rejection *contracts.RejectedOutput Rejection *contracts.RejectedOutput
Validation validationReport Validation validationReport
ValidationIncomplete bool ValidationIncomplete bool
@@ -121,7 +125,7 @@ type producerAttemptTerminal struct {
} }
func (terminal producerAttemptTerminal) clone() producerAttemptTerminal { func (terminal producerAttemptTerminal) clone() producerAttemptTerminal {
terminal.Warnings = cloneWarnings(terminal.Warnings) terminal.Diagnostics = contracts.CloneProducerDiagnostics(terminal.Diagnostics)
if terminal.Rejection != nil { if terminal.Rejection != nil {
rejection := *terminal.Rejection rejection := *terminal.Rejection
rejection.Validation = cloneValidationSummaryPtr(rejection.Validation) rejection.Validation = cloneValidationSummaryPtr(rejection.Validation)
@@ -193,6 +197,10 @@ func runProducerAttempts(ctx context.Context, config producerAttemptConfig, prod
} }
return failedProducerAttempt(provenance), fmt.Errorf("producer failed after %d attempt(s): %w", number, err) return failedProducerAttempt(provenance), fmt.Errorf("producer failed after %d attempt(s): %w", number, err)
} }
if err := validateProducerAttemptDiagnostics(output); err != nil {
provenance = append(provenance, producerAttemptProvenance{Number: number, Kind: kind, Outcome: producerAttemptFailed})
return failedProducerAttempt(provenance), err
}
output, err = output.clone() output, err = output.clone()
if err != nil { if err != nil {
@@ -200,12 +208,23 @@ func runProducerAttempts(ctx context.Context, config producerAttemptConfig, prod
return failedProducerAttempt(provenance), err return failedProducerAttempt(provenance), err
} }
if output.Retry != nil && number < attemptLimit { if output.Retry != nil && number < attemptLimit {
correction, err = moduleRetryCorrection(output)
if err != nil {
provenance = append(provenance, producerAttemptProvenance{Number: number, Kind: kind, Outcome: producerAttemptFailed})
return failedProducerAttempt(provenance), err
}
provenance = append(provenance, producerAttemptProvenance{Number: number, Kind: kind, Outcome: producerAttemptRetried}) provenance = append(provenance, producerAttemptProvenance{Number: number, Kind: kind, Outcome: producerAttemptRetried})
kind, correction = producerAttemptModuleRetry, nil kind = producerAttemptModuleRetry
continue continue
} }
if output.Retry != nil && output.Retry.CorrectionGuidance != "" {
if _, err := moduleRetryCorrection(output); err != nil {
provenance = append(provenance, producerAttemptProvenance{Number: number, Kind: kind, Outcome: producerAttemptFailed})
return failedProducerAttempt(provenance), err
}
}
if output.Retry != nil { if output.Retry != nil {
output.Warnings = append(output.Warnings, cloneWarnings(output.Retry.FallbackWarnings)...) output.Diagnostics = append(output.Diagnostics, contracts.CloneProducerDiagnostics(output.Retry.FallbackDiagnostics)...)
} }
correctionCandidate, err := contracts.CloneModelCandidate(output.Candidate) correctionCandidate, err := contracts.CloneModelCandidate(output.Candidate)
if err != nil { if err != nil {
@@ -253,20 +272,47 @@ func runProducerAttempts(ctx context.Context, config producerAttemptConfig, prod
if incomplete := firstIncompleteValidation(report); incomplete != nil { if incomplete := firstIncompleteValidation(report); incomplete != nil {
provenance = append(provenance, producerAttemptProvenance{Number: number, Kind: kind, Outcome: producerAttemptIncompleteAccepted, Validation: report}) provenance = append(provenance, producerAttemptProvenance{Number: number, Kind: kind, Outcome: producerAttemptIncompleteAccepted, Validation: report})
if config.Policy.ValidatorFailure == ValidatorFailureWarnContinue { if config.Policy.ValidatorFailure == ValidatorFailureWarnContinue {
warnings := terminalWarnings(output, report) return producerAttemptTerminal{Action: producerTerminalIncompleteAccepted, Value: output.Value, Diagnostics: cloneProducerDiagnostics(output.Diagnostics), Validation: report, ValidationIncomplete: true, Provenance: cloneProducerAttemptProvenance(provenance)}, nil
warnings = append(warnings, incompleteValidationWarnings(report)...)
return producerAttemptTerminal{Action: producerTerminalIncompleteAccepted, Value: output.Value, Warnings: warnings, Validation: report, ValidationIncomplete: true, Provenance: cloneProducerAttemptProvenance(provenance)}, nil
} }
return failedProducerAttempt(provenance), validatorFailureError(*incomplete) return failedProducerAttempt(provenance), validatorFailureError(*incomplete)
} }
provenance = append(provenance, producerAttemptProvenance{Number: number, Kind: kind, Outcome: producerAttemptAccepted, Validation: report}) provenance = append(provenance, producerAttemptProvenance{Number: number, Kind: kind, Outcome: producerAttemptAccepted, Validation: report})
return producerAttemptTerminal{Action: producerTerminalAccepted, Value: output.Value, Warnings: terminalWarnings(output, report), Validation: report, Provenance: cloneProducerAttemptProvenance(provenance)}, nil return producerAttemptTerminal{Action: producerTerminalAccepted, Value: output.Value, Diagnostics: cloneProducerDiagnostics(output.Diagnostics), Validation: report, Provenance: cloneProducerAttemptProvenance(provenance)}, nil
} }
return failedProducerAttempt(provenance), errors.New("producer attempt budget was not exhausted deterministically") return failedProducerAttempt(provenance), errors.New("producer attempt budget was not exhausted deterministically")
} }
func moduleRetryCorrection(output producerAttemptOutput) (*contracts.SemanticCorrection, error) {
if output.Retry == nil || output.Retry.CorrectionGuidance == "" {
return nil, nil
}
if output.Candidate == nil {
return nil, errors.New("module retry correction guidance requires a model candidate")
}
if output.Candidate.Protocol != contracts.CorrectionProtocolSingleResponseV1 {
return nil, fmt.Errorf("module retry correction guidance requires protocol %q", contracts.CorrectionProtocolSingleResponseV1)
}
correction, err := contracts.NewSemanticCorrection(output.Candidate.Response, output.Retry.CorrectionGuidance)
if err != nil {
return nil, fmt.Errorf("construct module retry semantic correction: %w", err)
}
return correction, nil
}
func validateProducerAttemptDiagnostics(output producerAttemptOutput) error {
if err := contracts.ValidateProducerDiagnostics(output.Diagnostics); err != nil {
return fmt.Errorf("producer returned invalid diagnostics: %w", err)
}
if output.Retry != nil {
if err := contracts.ValidateProducerDiagnostics(output.Retry.FallbackDiagnostics); err != nil {
return fmt.Errorf("producer returned invalid retry fallback diagnostics: %w", err)
}
}
return nil
}
func isImmediateProducerFailure(err error) bool { func isImmediateProducerFailure(err error) bool {
if errors.Is(err, context.Canceled) || errors.Is(err, context.DeadlineExceeded) { if errors.Is(err, context.Canceled) || errors.Is(err, context.DeadlineExceeded) {
return true return true
@@ -290,7 +336,7 @@ func applySemanticTerminalPolicy(policy ValidationPolicy, provenance []producerA
rejected := contracts.RejectedOutput{ValidatorName: rejection.validatorName, ReasonCode: rejection.reasonCode, Message: rejection.message, AttemptCount: number, DiagnosticArtifactPath: rejection.diagnosticPath} rejected := contracts.RejectedOutput{ValidatorName: rejection.validatorName, ReasonCode: rejection.reasonCode, Message: rejection.message, AttemptCount: number, DiagnosticArtifactPath: rejection.diagnosticPath}
switch policy.SemanticRejection { switch policy.SemanticRejection {
case SemanticRejectionRejectOutput: case SemanticRejectionRejectOutput:
return producerAttemptTerminal{Action: producerTerminalRejected, Warnings: terminalWarnings(output, report), Rejection: &rejected, Validation: report, ValidationIncomplete: firstIncompleteValidation(report) != nil, Provenance: cloneProducerAttemptProvenance(provenance)}, nil return producerAttemptTerminal{Action: producerTerminalRejected, Diagnostics: cloneProducerDiagnostics(output.Diagnostics), Rejection: &rejected, Validation: report, ValidationIncomplete: firstIncompleteValidation(report) != nil, Provenance: cloneProducerAttemptProvenance(provenance)}, nil
case SemanticRejectionFailRun: case SemanticRejectionFailRun:
return failedProducerAttempt(provenance), fmt.Errorf("producer candidate rejected after %d attempt(s): %s", number, rejection.message) return failedProducerAttempt(provenance), fmt.Errorf("producer candidate rejected after %d attempt(s): %s", number, rejection.message)
default: default:
@@ -308,28 +354,31 @@ func firstIncompleteValidation(report validationReport) *validationRecord {
return nil return nil
} }
func terminalWarnings(output producerAttemptOutput, report validationReport) []contracts.Warning { func cloneProducerDiagnostics(diagnostics []contracts.ProducerDiagnostic) []contracts.ProducerDiagnostic {
warnings := cloneWarnings(output.Warnings) return contracts.CloneProducerDiagnostics(diagnostics)
warnings = append(warnings, report.Warnings()...)
return warnings
} }
// incompleteValidationWarnings reports only validators that exhausted their // incompleteValidationDiagnostics reports every applicable validator that
// execution budget. It never reports rejected candidates, and it uses fixed // could not complete under warn_continue. It uses fixed text so provider
// text so provider errors and correction content cannot cross this boundary. // errors and arbitrary validator prose cannot cross this boundary.
func incompleteValidationWarnings(report validationReport) []contracts.Warning { func incompleteValidationDiagnostics(report validationReport) []validationDiagnosticRecord {
warnings := make([]contracts.Warning, 0) diagnostics := make([]validationDiagnosticRecord, 0)
for _, record := range report.records { for _, record := range report.records {
if record.outcome != validationFailed { if record.outcome != validationFailed && record.outcome != validationSkipped {
continue continue
} }
warnings = append(warnings, contracts.Warning{ diagnostics = append(diagnostics, validationDiagnosticRecord{validatorName: record.validatorName, diagnostic: contracts.ProducerDiagnostic{
Scope: record.validatorName, Disposition: contracts.DiagnosticDispositionWarning,
Category: contracts.DiagnosticCategoryValidationIncomplete,
ReasonCode: "validator_execution_incomplete", ReasonCode: "validator_execution_incomplete",
OccurrenceCount: 1,
Samples: []contracts.DiagnosticSample{{
Scope: record.validatorName,
Message: "Validator execution did not complete within its configured budget.", Message: "Validator execution did not complete within its configured budget.",
}) }},
}})
} }
return warnings return diagnostics
} }
// validationSummary projects a terminal state-machine result into the durable // validationSummary projects a terminal state-machine result into the durable

View File

@@ -4,6 +4,7 @@ import (
"context" "context"
"errors" "errors"
"reflect" "reflect"
"strings"
"testing" "testing"
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts" "gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
@@ -217,6 +218,9 @@ func TestRunProducerAttemptsUsesModuleRetryBudgetAndFallback(t *testing.T) {
calls := 0 calls := 0
terminal, err := runProducerAttempts(context.Background(), producerAttemptConfig{Retries: 1, Policy: DefaultValidationPolicy()}, func(_ context.Context, request producerAttemptRequest) (producerAttemptOutput, error) { terminal, err := runProducerAttempts(context.Background(), producerAttemptConfig{Retries: 1, Policy: DefaultValidationPolicy()}, func(_ context.Context, request producerAttemptRequest) (producerAttemptOutput, error) {
calls++ calls++
if request.Correction != nil {
t.Fatalf("feedback-free module retry correction = %#v, want nil", request.Correction)
}
if request.Number == 1 { if request.Number == 1 {
return producerAttemptOutput{Value: "fallback", Retry: &producerRetryDirective{}}, nil return producerAttemptOutput{Value: "fallback", Retry: &producerRetryDirective{}}, nil
} }
@@ -233,28 +237,137 @@ func TestRunProducerAttemptsUsesModuleRetryBudgetAndFallback(t *testing.T) {
} }
}) })
t.Run("fallback", func(t *testing.T) { t.Run("feedback retry", func(t *testing.T) {
fallbackWarning := contracts.Warning{ReasonCode: "fallback", Message: "fallback warning"} const (
terminal, err := runProducerAttempts(context.Background(), producerAttemptConfig{Policy: DefaultValidationPolicy()}, func(context.Context, producerAttemptRequest) (producerAttemptOutput, error) { defective = `{"duplicate_groups":[{"candidate_numbers":[1,99],"canonical_candidate_number":1}]}`
return producerAttemptOutput{Value: "fallback", Retry: &producerRetryDirective{FallbackWarnings: []contracts.Warning{fallbackWarning}}}, nil guidance = "Use only candidate numbers from the supplied candidate list. Return one complete corrected response."
)
var observed *contracts.SemanticCorrection
terminal, err := runProducerAttempts(context.Background(), producerAttemptConfig{Retries: 1, Policy: DefaultValidationPolicy()}, func(_ context.Context, request producerAttemptRequest) (producerAttemptOutput, error) {
if request.Number == 1 {
return producerAttemptOutput{Value: "safe fallback", Candidate: attemptCandidate(t, defective), Retry: &producerRetryDirective{CorrectionGuidance: guidance}}, nil
}
observed = request.Correction
return producerAttemptOutput{Value: "corrected"}, nil
}, approveAttempt) }, approveAttempt)
if err != nil { if err != nil {
t.Fatalf("runProducerAttempts() error = %v", err) t.Fatalf("runProducerAttempts() error = %v", err)
} }
if terminal.Action != producerTerminalAccepted || terminal.Value != "fallback" || !reflect.DeepEqual(terminal.Warnings, []contracts.Warning{fallbackWarning}) { if terminal.Action != producerTerminalAccepted || terminal.Value != "corrected" {
t.Fatalf("terminal = %#v, want corrected accepted value", terminal)
}
if observed == nil || string(observed.AssistantResponse) != defective || observed.UserGuidance != guidance {
t.Fatalf("module retry correction = %#v, want exact latest response and guidance", observed)
}
if got := attemptKinds(terminal.Provenance); !reflect.DeepEqual(got, []producerAttemptKind{producerAttemptInitial, producerAttemptModuleRetry}) {
t.Fatalf("attempt kinds = %v", got)
}
})
t.Run("feedback-free retry clears prior correction", func(t *testing.T) {
terminal, err := runProducerAttempts(context.Background(), producerAttemptConfig{Retries: 2, Policy: DefaultValidationPolicy()}, func(_ context.Context, request producerAttemptRequest) (producerAttemptOutput, error) {
switch request.Number {
case 1:
return producerAttemptOutput{Value: "first fallback", Candidate: attemptCandidate(t, "first defective response"), Retry: &producerRetryDirective{CorrectionGuidance: "Correct the first response."}}, nil
case 2:
if request.Correction == nil || string(request.Correction.AssistantResponse) != "first defective response" {
t.Fatalf("second attempt correction = %#v", request.Correction)
}
return producerAttemptOutput{Value: "second fallback", Retry: &producerRetryDirective{}}, nil
case 3:
if request.Correction != nil {
t.Fatalf("third attempt retained stale correction %#v", request.Correction)
}
return producerAttemptOutput{Value: "accepted"}, nil
default:
t.Fatalf("unexpected producer attempt %d", request.Number)
return producerAttemptOutput{}, nil
}
}, approveAttempt)
if err != nil || terminal.Action != producerTerminalAccepted || terminal.Value != "accepted" {
t.Fatalf("terminal = %#v, error = %v", terminal, err)
}
})
t.Run("fallback", func(t *testing.T) {
fallbackDiagnostic := contracts.ProducerDiagnostic{Disposition: contracts.DiagnosticDispositionWarning, Category: contracts.DiagnosticCategoryFallback, ReasonCode: "fallback", OccurrenceCount: 1, Samples: []contracts.DiagnosticSample{{Scope: "fallback", Message: "fallback warning"}}}
terminal, err := runProducerAttempts(context.Background(), producerAttemptConfig{Policy: DefaultValidationPolicy()}, func(context.Context, producerAttemptRequest) (producerAttemptOutput, error) {
return producerAttemptOutput{Value: "fallback", Retry: &producerRetryDirective{FallbackDiagnostics: []contracts.ProducerDiagnostic{fallbackDiagnostic}}}, nil
}, approveAttempt)
if err != nil {
t.Fatalf("runProducerAttempts() error = %v", err)
}
if terminal.Action != producerTerminalAccepted || terminal.Value != "fallback" || !reflect.DeepEqual(terminal.Diagnostics, []contracts.ProducerDiagnostic{fallbackDiagnostic}) {
t.Fatalf("terminal = %#v", terminal) t.Fatalf("terminal = %#v", terminal)
} }
}) })
} }
func TestRunProducerAttemptsRejectsModuleCorrectionWithoutModelCandidate(t *testing.T) {
producerCalls := 0
validatorCalls := 0
terminal, err := runProducerAttempts(context.Background(), producerAttemptConfig{Retries: 1, Policy: DefaultValidationPolicy()}, func(context.Context, producerAttemptRequest) (producerAttemptOutput, error) {
producerCalls++
return producerAttemptOutput{Value: "safe", Retry: &producerRetryDirective{CorrectionGuidance: "Return a complete corrected response."}}, nil
}, func(context.Context, producerAttemptOutput) (validationReport, error) {
validatorCalls++
return validationReport{}, nil
})
if err == nil || !strings.Contains(err.Error(), "requires a model candidate") {
t.Fatalf("runProducerAttempts() error = %v, want model-candidate contract failure", err)
}
if terminal.Action != producerTerminalFailed || producerCalls != 1 || validatorCalls != 0 {
t.Fatalf("terminal = %#v, producer calls = %d, validator calls = %d", terminal, producerCalls, validatorCalls)
}
}
func TestRunProducerAttemptsRejectsInvalidDiagnosticsWithoutRetry(t *testing.T) {
tests := []struct {
name string
output producerAttemptOutput
want string
}{
{
name: "producer diagnostics",
output: producerAttemptOutput{Diagnostics: []contracts.ProducerDiagnostic{{}}},
want: "producer returned invalid diagnostics",
},
{
name: "retry fallback diagnostics",
output: producerAttemptOutput{Retry: &producerRetryDirective{FallbackDiagnostics: []contracts.ProducerDiagnostic{{}}}},
want: "producer returned invalid retry fallback diagnostics",
},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
producerCalls := 0
validatorCalls := 0
terminal, err := runProducerAttempts(context.Background(), producerAttemptConfig{Retries: 3, Policy: DefaultValidationPolicy()}, func(context.Context, producerAttemptRequest) (producerAttemptOutput, error) {
producerCalls++
return test.output, nil
}, func(context.Context, producerAttemptOutput) (validationReport, error) {
validatorCalls++
return validationReport{}, nil
})
if err == nil || !strings.Contains(err.Error(), test.want) {
t.Fatalf("runProducerAttempts() error = %v, want %q", err, test.want)
}
if terminal.Action != producerTerminalFailed || producerCalls != 1 || validatorCalls != 0 {
t.Fatalf("terminal = %#v, producer calls = %d, validator calls = %d; want immediate framework failure", terminal, producerCalls, validatorCalls)
}
})
}
}
func TestRunProducerAttemptsRejectionWinsOverValidatorFailure(t *testing.T) { func TestRunProducerAttemptsRejectionWinsOverValidatorFailure(t *testing.T) {
firstWarnings := []contracts.Warning{{ReasonCode: "discarded", Message: "discarded warning"}} firstDiagnostics := []contracts.ProducerDiagnostic{producerDiagnostic("discarded", "discarded warning")}
secondWarnings := []contracts.Warning{{ReasonCode: "accepted", Message: "accepted warning"}} secondDiagnostics := []contracts.ProducerDiagnostic{producerDiagnostic("accepted", "accepted warning")}
terminal, err := runProducerAttempts(context.Background(), producerAttemptConfig{Retries: 1, Policy: DefaultValidationPolicy()}, func(_ context.Context, request producerAttemptRequest) (producerAttemptOutput, error) { terminal, err := runProducerAttempts(context.Background(), producerAttemptConfig{Retries: 1, Policy: DefaultValidationPolicy()}, func(_ context.Context, request producerAttemptRequest) (producerAttemptOutput, error) {
if request.Number == 1 { if request.Number == 1 {
return producerAttemptOutput{Value: "first", Candidate: attemptCandidate(t, "defective"), Warnings: firstWarnings}, nil return producerAttemptOutput{Value: "first", Candidate: attemptCandidate(t, "defective"), Diagnostics: firstDiagnostics}, nil
} }
return producerAttemptOutput{Value: "second", Candidate: attemptCandidate(t, "corrected"), Warnings: secondWarnings}, nil return producerAttemptOutput{Value: "second", Candidate: attemptCandidate(t, "corrected"), Diagnostics: secondDiagnostics}, nil
}, func(_ context.Context, output producerAttemptOutput) (validationReport, error) { }, func(_ context.Context, output producerAttemptOutput) (validationReport, error) {
if output.Value == "first" { if output.Value == "first" {
report := rejectedAttemptReport("defect") report := rejectedAttemptReport("defect")
@@ -272,8 +385,8 @@ func TestRunProducerAttemptsRejectionWinsOverValidatorFailure(t *testing.T) {
if got := attemptKinds(terminal.Provenance); !reflect.DeepEqual(got, []producerAttemptKind{producerAttemptInitial, producerAttemptSemanticRetry}) { if got := attemptKinds(terminal.Provenance); !reflect.DeepEqual(got, []producerAttemptKind{producerAttemptInitial, producerAttemptSemanticRetry}) {
t.Fatalf("attempt kinds = %v", got) t.Fatalf("attempt kinds = %v", got)
} }
if got := terminal.Warnings; !reflect.DeepEqual(got, secondWarnings) { if got := terminal.Diagnostics; !reflect.DeepEqual(got, secondDiagnostics) {
t.Fatalf("terminal warnings = %#v, want %#v", got, secondWarnings) t.Fatalf("terminal diagnostics = %#v, want %#v", got, secondDiagnostics)
} }
if len(terminal.Provenance[0].Validation.records) != 2 { if len(terminal.Provenance[0].Validation.records) != 2 {
t.Fatalf("first validation records = %#v, want rejection and failure", terminal.Provenance[0].Validation.records) t.Fatalf("first validation records = %#v, want rejection and failure", terminal.Provenance[0].Validation.records)
@@ -302,9 +415,6 @@ func TestValidationSummaryIsBoundedAndCorrectedSuccessIsQuiet(t *testing.T) {
if len(summary.ReasonCodes) != 0 || len(summary.RejectingValidators) != 0 || len(summary.IncompleteValidators) != 0 { if len(summary.ReasonCodes) != 0 || len(summary.RejectingValidators) != 0 || len(summary.IncompleteValidators) != 0 {
t.Fatalf("corrected success summary retained prior findings: %#v", summary) t.Fatalf("corrected success summary retained prior findings: %#v", summary)
} }
if len(terminal.Warnings) != 0 {
t.Fatalf("corrected success warnings = %#v, want none", terminal.Warnings)
}
} }
func TestWarnContinueRecordsOneWarningForEachExhaustedValidator(t *testing.T) { func TestWarnContinueRecordsOneWarningForEachExhaustedValidator(t *testing.T) {
@@ -324,11 +434,17 @@ func TestWarnContinueRecordsOneWarningForEachExhaustedValidator(t *testing.T) {
if terminal.Action != producerTerminalIncompleteAccepted { if terminal.Action != producerTerminalIncompleteAccepted {
t.Fatalf("terminal action = %q", terminal.Action) t.Fatalf("terminal action = %q", terminal.Action)
} }
if got, want := terminal.Warnings, []contracts.Warning{ groups, groupErr := terminalDiagnosticGroups(terminal, contracts.DiagnosticOrigin{Stage: contracts.DiagnosticOriginStageNormalize, StepID: "step", LaneID: "lane", ModuleKey: "module"}, nil)
{Scope: "first", ReasonCode: "validator_execution_incomplete", Message: "Validator execution did not complete within its configured budget."}, if groupErr != nil {
{Scope: "third", ReasonCode: "validator_execution_incomplete", Message: "Validator execution did not complete within its configured budget."}, t.Fatalf("terminalDiagnosticGroups() error = %v", groupErr)
}; !reflect.DeepEqual(got, want) { }
t.Fatalf("warnings = %#v, want %#v", got, want) if len(groups) != 3 {
t.Fatalf("diagnostic groups = %#v, want every failed and skipped validator", groups)
}
for index, validator := range []string{"first", "second", "third"} {
if group := groups[index]; group.Origin.ValidatorKey != validator || group.Disposition != contracts.DiagnosticDispositionWarning || group.Category != contracts.DiagnosticCategoryValidationIncomplete || group.ReasonCode != "validator_execution_incomplete" {
t.Fatalf("diagnostic group %d = %#v, want incomplete warning for %q", index, group, validator)
}
} }
summary := validationSummary(terminal, StageNormalize, "step", "lane", "module", "", 0) summary := validationSummary(terminal, StageNormalize, "step", "lane", "module", "", 0)
if summary.Status != "incomplete" || !reflect.DeepEqual(summary.IncompleteValidators, []string{"first", "second", "third"}) || !reflect.DeepEqual(summary.ReasonCodes, []string{"missing_prerequisite"}) { if summary.Status != "incomplete" || !reflect.DeepEqual(summary.IncompleteValidators, []string{"first", "second", "third"}) || !reflect.DeepEqual(summary.ReasonCodes, []string{"missing_prerequisite"}) {

View File

@@ -30,45 +30,45 @@ type ReferenceMaterializationOptions struct {
WorkingDir string WorkingDir string
} }
func MaterializeReferences(resolved ResolvedPipeline, catalog ModuleCatalog, options ReferenceMaterializationOptions) (ResolvedPipeline, []contracts.Warning, error) { func MaterializeReferences(resolved ResolvedPipeline, catalog ModuleCatalog, options ReferenceMaterializationOptions) (ResolvedPipeline, []contracts.ProducerDiagnostic, error) {
out := resolved out := resolved
out.ChunkReferences = CloneReferenceTarget(resolved.ChunkReferences) out.ChunkReferences = CloneReferenceTarget(resolved.ChunkReferences)
chunkReferenceSet, chunkWarnings, err := materializeReferenceTarget(resolved.ID, resolved.ChunkReferences, "", catalog, options) chunkReferenceSet, chunkDiagnostics, err := materializeReferenceTarget(resolved.ID, resolved.ChunkReferences, "", catalog, options)
if err != nil { if err != nil {
return ResolvedPipeline{}, nil, err return ResolvedPipeline{}, nil, err
} }
out.ChunkReferences.ReferenceSet = chunkReferenceSet out.ChunkReferences.ReferenceSet = chunkReferenceSet
warnings := append([]contracts.Warning(nil), chunkWarnings...) diagnostics := contracts.CloneProducerDiagnostics(chunkDiagnostics)
if len(resolved.Steps) == 0 { if len(resolved.Steps) == 0 {
return out, warnings, nil return out, diagnostics, nil
} }
materializeLane := func(lane ResolvedArtifactLane) (ResolvedArtifactLane, []contracts.Warning, error) { materializeLane := func(lane ResolvedArtifactLane) (ResolvedArtifactLane, []contracts.ProducerDiagnostic, error) {
materializedLane := lane materializedLane := lane
var allWarnings []contracts.Warning var allDiagnostics []contracts.ProducerDiagnostic
materializedLane.ExtractReferences = CloneReferenceTarget(lane.ExtractReferences) materializedLane.ExtractReferences = CloneReferenceTarget(lane.ExtractReferences)
materializedLane.MergeReferences = CloneReferenceTarget(lane.MergeReferences) materializedLane.MergeReferences = CloneReferenceTarget(lane.MergeReferences)
materializedLane.NormalizeReferences = CloneReferenceTarget(lane.NormalizeReferences) materializedLane.NormalizeReferences = CloneReferenceTarget(lane.NormalizeReferences)
extractReferenceSet, laneWarnings, err := materializeReferenceTarget(resolved.ID, lane.ExtractReferences, lane.ArtifactKind, catalog, options) extractReferenceSet, laneDiagnostics, err := materializeReferenceTarget(resolved.ID, lane.ExtractReferences, lane.ArtifactKind, catalog, options)
if err != nil { if err != nil {
return ResolvedArtifactLane{}, nil, err return ResolvedArtifactLane{}, nil, err
} }
materializedLane.ExtractReferences.ReferenceSet = extractReferenceSet materializedLane.ExtractReferences.ReferenceSet = extractReferenceSet
allWarnings = append(allWarnings, laneWarnings...) allDiagnostics = append(allDiagnostics, laneDiagnostics...)
mergeReferenceSet, laneWarnings, err := materializeReferenceTarget(resolved.ID, lane.MergeReferences, lane.ArtifactKind, catalog, options) mergeReferenceSet, laneDiagnostics, err := materializeReferenceTarget(resolved.ID, lane.MergeReferences, lane.ArtifactKind, catalog, options)
if err != nil { if err != nil {
return ResolvedArtifactLane{}, nil, err return ResolvedArtifactLane{}, nil, err
} }
materializedLane.MergeReferences.ReferenceSet = mergeReferenceSet materializedLane.MergeReferences.ReferenceSet = mergeReferenceSet
allWarnings = append(allWarnings, laneWarnings...) allDiagnostics = append(allDiagnostics, laneDiagnostics...)
normalizeReferenceSet, laneWarnings, err := materializeReferenceTarget(resolved.ID, lane.NormalizeReferences, lane.ArtifactKind, catalog, options) normalizeReferenceSet, laneDiagnostics, err := materializeReferenceTarget(resolved.ID, lane.NormalizeReferences, lane.ArtifactKind, catalog, options)
if err != nil { if err != nil {
return ResolvedArtifactLane{}, nil, err return ResolvedArtifactLane{}, nil, err
} }
materializedLane.NormalizeReferences.ReferenceSet = normalizeReferenceSet materializedLane.NormalizeReferences.ReferenceSet = normalizeReferenceSet
allWarnings = append(allWarnings, laneWarnings...) allDiagnostics = append(allDiagnostics, laneDiagnostics...)
return materializedLane, allWarnings, nil return materializedLane, allDiagnostics, nil
} }
if len(resolved.Steps) > 0 { if len(resolved.Steps) > 0 {
out.Steps = make([]ResolvedPipelineStep, len(resolved.Steps)) out.Steps = make([]ResolvedPipelineStep, len(resolved.Steps))
@@ -76,16 +76,16 @@ func MaterializeReferences(resolved ResolvedPipeline, catalog ModuleCatalog, opt
out.Steps[i].ID = step.ID out.Steps[i].ID = step.ID
out.Steps[i].ArtifactLanes = make([]ResolvedArtifactLane, len(step.ArtifactLanes)) out.Steps[i].ArtifactLanes = make([]ResolvedArtifactLane, len(step.ArtifactLanes))
for j, lane := range step.ArtifactLanes { for j, lane := range step.ArtifactLanes {
materializedLane, laneWarnings, err := materializeLane(lane) materializedLane, laneDiagnostics, err := materializeLane(lane)
if err != nil { if err != nil {
return ResolvedPipeline{}, nil, err return ResolvedPipeline{}, nil, err
} }
out.Steps[i].ArtifactLanes[j] = materializedLane out.Steps[i].ArtifactLanes[j] = materializedLane
warnings = append(warnings, laneWarnings...) diagnostics = append(diagnostics, laneDiagnostics...)
} }
} }
} }
return out, warnings, nil return out, diagnostics, nil
} }
func materializeReferenceTarget( func materializeReferenceTarget(
@@ -94,7 +94,7 @@ func materializeReferenceTarget(
artifactKind contracts.ArtifactKind, artifactKind contracts.ArtifactKind,
catalog ModuleCatalog, catalog ModuleCatalog,
options ReferenceMaterializationOptions, options ReferenceMaterializationOptions,
) (contracts.ReferenceSet, []contracts.Warning, error) { ) (contracts.ReferenceSet, []contracts.ProducerDiagnostic, error) {
if len(target.Bindings) == 0 { if len(target.Bindings) == 0 {
return contracts.ReferenceSet{}, nil, nil return contracts.ReferenceSet{}, nil, nil
} }
@@ -109,7 +109,7 @@ func materializeReferenceTarget(
} }
set := contracts.ReferenceSet{Slots: make(map[string]contracts.ResolvedReferenceSlot, len(target.Bindings))} set := contracts.ReferenceSet{Slots: make(map[string]contracts.ResolvedReferenceSlot, len(target.Bindings))}
var warnings []contracts.Warning var diagnostics []contracts.ProducerDiagnostic
for _, binding := range target.Bindings { for _, binding := range target.Bindings {
slotName := strings.TrimSpace(binding.SlotName) slotName := strings.TrimSpace(binding.SlotName)
slot, ok := slotByName[slotName] slot, ok := slotByName[slotName]
@@ -159,11 +159,7 @@ func materializeReferenceTarget(
return contracts.ReferenceSet{}, nil, fmt.Errorf("%s reference slot %q path %q media type %q is not accepted", referenceTargetContext(pipelineID, target), slotName, path, mediaType) return contracts.ReferenceSet{}, nil, fmt.Errorf("%s reference slot %q path %q media type %q is not accepted", referenceTargetContext(pipelineID, target), slotName, path, mediaType)
} }
if len(content) == 0 { if len(content) == 0 {
warnings = append(warnings, contracts.Warning{ diagnostics = append(diagnostics, contracts.ProducerDiagnostic{Disposition: contracts.DiagnosticDispositionWarning, Category: contracts.DiagnosticCategoryConfiguration, ReasonCode: "empty_reference", OccurrenceCount: 1, Samples: []contracts.DiagnosticSample{{Scope: referenceWarningScope(pipelineID, target, slotName), Message: fmt.Sprintf("reference slot %q for %s is bound to an empty file", slotName, referenceTargetLabel(target))}}})
Scope: referenceWarningScope(pipelineID, target, slotName),
ReasonCode: "empty_reference",
Message: fmt.Sprintf("reference slot %q for %s is bound to an empty file", slotName, referenceTargetLabel(target)),
})
} }
item := contracts.ReferenceItem{ item := contracts.ReferenceItem{
@@ -180,7 +176,7 @@ func materializeReferenceTarget(
Items: []contracts.ReferenceItem{item}, Items: []contracts.ReferenceItem{item},
} }
} }
return set, warnings, nil return set, diagnostics, nil
} }
type referenceSizeLimitError struct { type referenceSizeLimitError struct {

View File

@@ -445,14 +445,14 @@ func TestMaterializeReferencesWarnsForEmptyFiles(t *testing.T) {
writeReferenceFile(t, path, nil) writeReferenceFile(t, path, nil)
resolved := resolvedPipelineWithReference(t, "roster", "empty.txt", contracts.ReferenceBindingSourceConfig, contracts.ReferenceSlot{Name: "roster"}) resolved := resolvedPipelineWithReference(t, "roster", "empty.txt", contracts.ReferenceBindingSourceConfig, contracts.ReferenceSlot{Name: "roster"})
materialized, warnings, err := MaterializeReferences(resolved, referenceCatalog(t, []contracts.ReferenceSlot{{Name: "roster"}}), ReferenceMaterializationOptions{ materialized, diagnostics, err := MaterializeReferences(resolved, referenceCatalog(t, []contracts.ReferenceSlot{{Name: "roster"}}), ReferenceMaterializationOptions{
ConfigPath: filepath.Join(configDir, "config.yml"), ConfigPath: filepath.Join(configDir, "config.yml"),
}) })
if err != nil { if err != nil {
t.Fatalf("MaterializeReferences() error = %v, want nil", err) t.Fatalf("MaterializeReferences() error = %v, want nil", err)
} }
if len(warnings) != 1 || warnings[0].ReasonCode != "empty_reference" { if len(diagnostics) != 1 || diagnostics[0].Disposition != contracts.DiagnosticDispositionWarning || diagnostics[0].Category != contracts.DiagnosticCategoryConfiguration || diagnostics[0].ReasonCode != "empty_reference" || diagnostics[0].OccurrenceCount != 1 || len(diagnostics[0].Samples) != 1 {
t.Fatalf("warnings = %#v, want empty reference warning", warnings) t.Fatalf("diagnostics = %#v, want structured empty-reference signal", diagnostics)
} }
item := materialized.Steps[0].ArtifactLanes[0].ExtractReferences.ReferenceSet.Slots["roster"].Items[0] item := materialized.Steps[0].ArtifactLanes[0].ExtractReferences.ReferenceSet.Slots["roster"].Items[0]
if item.SizeBytes != 0 || item.Digest != referenceDigest(nil) { if item.SizeBytes != 0 || item.Digest != referenceDigest(nil) {
@@ -483,13 +483,13 @@ func TestMaterializeReferencesWarningScopesIncludeTargetContext(t *testing.T) {
t.Fatalf("ResolvePipeline() error = %v, want nil", err) t.Fatalf("ResolvePipeline() error = %v, want nil", err)
} }
_, warnings, err := MaterializeReferences(resolved, catalog, ReferenceMaterializationOptions{ _, diagnostics, err := MaterializeReferences(resolved, catalog, ReferenceMaterializationOptions{
ConfigPath: filepath.Join(configDir, "config.yml"), ConfigPath: filepath.Join(configDir, "config.yml"),
}) })
if err != nil { if err != nil {
t.Fatalf("MaterializeReferences() error = %v, want nil", err) t.Fatalf("MaterializeReferences() error = %v, want nil", err)
} }
got := warningScopes(warnings) got := diagnosticScopes(diagnostics)
want := []string{ want := []string{
"pipeline.baseline.chunk.reference.scene_guide", "pipeline.baseline.chunk.reference.scene_guide",
"pipeline.baseline.lane.events.extract.reference.roster", "pipeline.baseline.lane.events.extract.reference.roster",
@@ -621,10 +621,10 @@ func referenceCatalogForTargets(t *testing.T, chunkSlots, extractSlots, mergeSlo
) )
} }
func warningScopes(warnings []contracts.Warning) []string { func diagnosticScopes(diagnostics []contracts.ProducerDiagnostic) []string {
scopes := make([]string, 0, len(warnings)) scopes := make([]string, 0, len(diagnostics))
for _, warning := range warnings { for _, diagnostic := range diagnostics {
scopes = append(scopes, warning.Scope) scopes = append(scopes, diagnostic.Samples[0].Scope)
} }
sort.Strings(scopes) sort.Strings(scopes)
return scopes return scopes

View File

@@ -50,7 +50,7 @@ type RunInput struct {
StartedAt time.Time StartedAt time.Time
LLMProfiles []artifacts.LLMProfileManifest LLMProfiles []artifacts.LLMProfileManifest
Metadata map[string]any Metadata map[string]any
Warnings []contracts.Warning Diagnostics []contracts.ProducerDiagnostic
ChunkCacheMode ChunkCacheMode ChunkCacheMode ChunkCacheMode
ChunkPlans ChunkPlanStore ChunkPlans ChunkPlanStore
Checkpoints CheckpointRecorder Checkpoints CheckpointRecorder
@@ -73,12 +73,13 @@ type RunOutput struct {
ChunkPlan *artifacts.ChunkPlanSummary `json:"chunk_plan,omitempty"` ChunkPlan *artifacts.ChunkPlanSummary `json:"chunk_plan,omitempty"`
NormalizeOutputs []contracts.SerializedOutput `json:"normalize_outputs,omitempty"` NormalizeOutputs []contracts.SerializedOutput `json:"normalize_outputs,omitempty"`
Rejected []contracts.RejectedOutput `json:"rejected,omitempty"` Rejected []contracts.RejectedOutput `json:"rejected,omitempty"`
Warnings []contracts.Warning `json:"warnings,omitempty"` Diagnostics contracts.DiagnosticCollection `json:"diagnostics,omitempty"`
OutputFiles []contracts.OutputFile `json:"-"` OutputFiles []contracts.OutputFile `json:"-"`
CheckpointEvents []CheckpointEvent `json:"checkpoint_events,omitempty"` CheckpointEvents []CheckpointEvent `json:"checkpoint_events,omitempty"`
ValidationSummaries []artifacts.ValidationSummary `json:"validation_summaries,omitempty"` ValidationSummaries []artifacts.ValidationSummary `json:"validation_summaries,omitempty"`
normalizeReuseEligibility map[generatedOutputKey]bool normalizeReuseEligibility map[generatedOutputKey]bool
diagnosticGroups []contracts.DiagnosticGroup
} }
func (r *Runner) Run(ctx context.Context, input RunInput) (output RunOutput, err error) { func (r *Runner) Run(ctx context.Context, input RunInput) (output RunOutput, err error) {
@@ -129,7 +130,11 @@ func (r *Runner) Run(ctx context.Context, input RunInput) (output RunOutput, err
defer func() { defer func() {
output.Manifest.LLMProfiles = mergeLLMProfileManifests(input.LLMProfiles, llmProfileManifests(input.llmClient)) output.Manifest.LLMProfiles = mergeLLMProfileManifests(input.LLMProfiles, llmProfileManifests(input.llmClient))
}() }()
output.Warnings = append(output.Warnings, cloneWarnings(input.Warnings)...) inputDiagnostics, diagnosticErr := promoteProducerDiagnostics(input.Diagnostics, contracts.DiagnosticOrigin{Stage: contracts.DiagnosticOriginStageReferences}, nil)
if diagnosticErr != nil {
return failOutput(output), fmt.Errorf("promote input diagnostics: %w", diagnosticErr)
}
appendDiagnosticGroups(&output, inputDiagnostics)
if err := writeDebugTimed(debugRecorder, "run.json", debugTimedEnvelope{ if err := writeDebugTimed(debugRecorder, "run.json", debugTimedEnvelope{
Stage: "run", Stage: "run",
StartedAt: startedTime(input.StartedAt), StartedAt: startedTime(input.StartedAt),
@@ -248,13 +253,12 @@ func (r *Runner) Run(ctx context.Context, input RunInput) (output RunOutput, err
if chunkResult.validation != nil { if chunkResult.validation != nil {
output.ValidationSummaries = append(output.ValidationSummaries, artifacts.CloneValidationSummary(*chunkResult.validation)) output.ValidationSummaries = append(output.ValidationSummaries, artifacts.CloneValidationSummary(*chunkResult.validation))
} }
output.Warnings = append(output.Warnings, chunkResult.warnings...) appendDiagnosticGroups(&output, chunkResult.diagnostics)
chunkDebugPayload := map[string]any{ chunkDebugPayload := map[string]any{
"cache_mode": chunkMode, "cache_mode": chunkMode,
"lookup": chunkResult.lookup, "lookup": chunkResult.lookup,
"accepted": chunkResult.accepted, "accepted": chunkResult.accepted,
"materialized_chunks": debugSourceChunkEnvelopes(chunkResult.chunks), "materialized_chunks": debugSourceChunkEnvelopes(chunkResult.chunks),
"warnings": chunkResult.warnings,
} }
if chunkResult.plan != nil { if chunkResult.plan != nil {
chunkDebugPayload["plan"] = debugChunkPlanEnvelope(*chunkResult.plan) chunkDebugPayload["plan"] = debugChunkPlanEnvelope(*chunkResult.plan)
@@ -311,6 +315,9 @@ func (r *Runner) Run(ctx context.Context, input RunInput) (output RunOutput, err
} }
populateOutputManifest(&output) populateOutputManifest(&output)
output.Manifest.CompletedAt = timePtr(time.Now().UTC()) output.Manifest.CompletedAt = timePtr(time.Now().UTC())
if err := finalizeDiagnostics(&output); err != nil {
return failOutput(output), fmt.Errorf("aggregate diagnostics: %w", err)
}
encoder := input.Prepared.output encoder := input.Prepared.output
if err := attachModuleManifestMetadata(&output, "output", encoder); err != nil { if err := attachModuleManifestMetadata(&output, "output", encoder); err != nil {
@@ -344,7 +351,6 @@ func (r *Runner) Run(ctx context.Context, input RunInput) (output RunOutput, err
"manifest": output.Manifest, "manifest": output.Manifest,
"normalize_outputs": debugSerializedOutputEnvelopes(output.NormalizeOutputs), "normalize_outputs": debugSerializedOutputEnvelopes(output.NormalizeOutputs),
"rejected": debugRejectedOutputEnvelopes(output.Rejected), "rejected": debugRejectedOutputEnvelopes(output.Rejected),
"warnings": output.Warnings,
"options": redactSensitiveMap(input.pipeline.Output.Options), "options": redactSensitiveMap(input.pipeline.Output.Options),
"metadata": redactSensitiveMap(input.Metadata), "metadata": redactSensitiveMap(input.Metadata),
} }
@@ -373,7 +379,7 @@ func (r *Runner) Run(ctx context.Context, input RunInput) (output RunOutput, err
Manifest: output.Manifest, Manifest: output.Manifest,
NormalizeOutputs: cloneSerializedOutputs(output.NormalizeOutputs), NormalizeOutputs: cloneSerializedOutputs(output.NormalizeOutputs),
Rejected: cloneRejectedOutputs(output.Rejected), Rejected: cloneRejectedOutputs(output.Rejected),
Warnings: output.Warnings, Diagnostics: contracts.CloneDiagnosticCollection(output.Diagnostics),
LLMProfile: input.pipeline.Output.LLMProfile, LLMProfile: input.pipeline.Output.LLMProfile,
StructuredOutputRepairAttempts: cloneStructuredOutputRepairAttempts(input.pipeline.Output.StructuredOutputRepairAttempts), StructuredOutputRepairAttempts: cloneStructuredOutputRepairAttempts(input.pipeline.Output.StructuredOutputRepairAttempts),
Metadata: outputMetadata, Metadata: outputMetadata,
@@ -396,7 +402,6 @@ func (r *Runner) Run(ctx context.Context, input RunInput) (output RunOutput, err
StartedAt: outputStarted, StartedAt: outputStarted,
Payload: map[string]any{ Payload: map[string]any{
"files": debugOutputFiles(files), "files": debugOutputFiles(files),
"warnings": encoded.Warnings,
}, },
}); err != nil { }); err != nil {
return failOutput(output), fmt.Errorf("write output debug artifact: %w", err) return failOutput(output), fmt.Errorf("write output debug artifact: %w", err)
@@ -404,7 +409,6 @@ func (r *Runner) Run(ctx context.Context, input RunInput) (output RunOutput, err
if err := ctx.Err(); err != nil { if err := ctx.Err(); err != nil {
return failOutput(output), err return failOutput(output), err
} }
output.Warnings = append(output.Warnings, encoded.Warnings...)
output.OutputFiles = files output.OutputFiles = files
return output, nil return output, nil
@@ -444,7 +448,6 @@ func (r *Runner) runPreparedSteps(ctx context.Context, input RunInput, checkpoin
type retryAttemptResult struct { type retryAttemptResult struct {
accepted bool accepted bool
rejection *contracts.RejectedOutput rejection *contracts.RejectedOutput
warnings []contracts.Warning
} }
func runSimpleRetry(ctx context.Context, retries int, run func(attempt int) (retryAttemptResult, error)) (retryAttemptResult, error) { func runSimpleRetry(ctx context.Context, retries int, run func(attempt int) (retryAttemptResult, error)) (retryAttemptResult, error) {
@@ -474,7 +477,7 @@ func runSimpleRetry(ctx context.Context, retries int, run func(attempt int) (ret
if result.rejection != nil { if result.rejection != nil {
rejection := *result.rejection rejection := *result.rejection
rejection.AttemptCount = attempt rejection.AttemptCount = attempt
last = retryAttemptResult{rejection: &rejection, warnings: cloneWarnings(result.warnings)} last = retryAttemptResult{rejection: &rejection}
} }
if ctxErr := ctx.Err(); ctxErr != nil { if ctxErr := ctx.Err(); ctxErr != nil {
return retryAttemptResult{}, ctxErr return retryAttemptResult{}, ctxErr
@@ -697,6 +700,10 @@ func validatorChainManifests(chains []ResolvedValidatorChain) []artifacts.Valida
} }
func failOutput(output RunOutput) RunOutput { func failOutput(output RunOutput) RunOutput {
// Preserve any diagnostics already accepted by the pipeline when a later
// operation fails. Producers and validators validate each diagnostic before
// it is appended, so finalization here cannot introduce a new failure path.
_ = finalizeDiagnostics(&output)
if output.Manifest.PipelineID != "" { if output.Manifest.PipelineID != "" {
populateOutputManifest(&output) populateOutputManifest(&output)
output.Manifest.ValidationStatus = "failed" output.Manifest.ValidationStatus = "failed"
@@ -971,13 +978,6 @@ func manifestMetadataWithSessionID(metadata map[string]any, sessionID string) (m
return out, nil return out, nil
} }
func cloneWarnings(warnings []contracts.Warning) []contracts.Warning {
if len(warnings) == 0 {
return nil
}
return append([]contracts.Warning(nil), warnings...)
}
func cloneSourceChunkPtr(chunk *source.Chunk) (*source.Chunk, error) { func cloneSourceChunkPtr(chunk *source.Chunk) (*source.Chunk, error) {
if chunk == nil { if chunk == nil {
return nil, nil return nil, nil

View File

@@ -84,7 +84,8 @@ func TestRunnerHydratesRequiredNormalizedArtifact(t *testing.T) {
loader := newAcceptedCheckpointLoader() loader := newAcceptedCheckpointLoader()
producerKey := CheckpointLaneKey(producer.resolved.StepID, producer.resolved.ID) producerKey := CheckpointLaneKey(producer.resolved.StepID, producer.resolved.ID)
consumerKey := CheckpointLaneKey(consumer.resolved.StepID, consumer.resolved.ID) consumerKey := CheckpointLaneKey(consumer.resolved.StepID, consumer.resolved.ID)
loader.accepted[producerKey] = NormalizeCheckpoint{Output: stored, Warnings: []contracts.Warning{{Scope: "normalize", ReasonCode: "stored-warning", Message: "stored normalize warning"}}} stored.Diagnostics = []CheckpointDiagnostic{{Diagnostic: contracts.ProducerDiagnostic{Disposition: contracts.DiagnosticDispositionWarning, Category: contracts.DiagnosticCategoryFallback, ReasonCode: "stored-warning", OccurrenceCount: 1, Samples: []contracts.DiagnosticSample{{Scope: "normalize", Message: "stored normalize warning"}}}}}
loader.accepted[producerKey] = NormalizeCheckpoint{Output: stored}
loader.acceptedDecision[producerKey] = NewCheckpointDecision(CheckpointDecisionReused, CheckpointReasonAcceptedArtifactReused) loader.acceptedDecision[producerKey] = NewCheckpointDecision(CheckpointDecisionReused, CheckpointReasonAcceptedArtifactReused)
policy := CheckpointExecutionPolicy{ policy := CheckpointExecutionPolicy{
RequireReusableLanes: map[string]struct{}{producerKey: {}}, RequireReusableLanes: map[string]struct{}{producerKey: {}},
@@ -101,8 +102,8 @@ func TestRunnerHydratesRequiredNormalizedArtifact(t *testing.T) {
if string(item.Content) != string(stored.Artifact.Content) || item.Producer.StepID != producer.resolved.StepID || item.Producer.LaneID != producer.resolved.ID { if string(item.Content) != string(stored.Artifact.Content) || item.Producer.StepID != producer.resolved.StepID || item.Producer.LaneID != producer.resolved.ID {
t.Fatalf("consumer generated reference = %#v, want exact hydrated producer bytes and identity", item) t.Fatalf("consumer generated reference = %#v, want exact hydrated producer bytes and identity", item)
} }
if len(output.Warnings) != 1 || output.Warnings[0].ReasonCode != "stored-warning" { if len(output.Diagnostics.Groups) != 1 || output.Diagnostics.Groups[0].ReasonCode != "stored-warning" {
t.Fatalf("hydrated warnings = %#v, want normalize checkpoint warnings only", output.Warnings) t.Fatalf("hydrated diagnostics = %#v, want normalize checkpoint diagnostics only", output.Diagnostics)
} }
assertAcceptedNormalizeEvent(t, output.CheckpointEvents, producer.resolved.StepID, producer.resolved.ID, CheckpointDecisionReused, CheckpointReasonAcceptedArtifactReused) assertAcceptedNormalizeEvent(t, output.CheckpointEvents, producer.resolved.StepID, producer.resolved.ID, CheckpointDecisionReused, CheckpointReasonAcceptedArtifactReused)
for _, event := range output.CheckpointEvents { for _, event := range output.CheckpointEvents {
@@ -225,7 +226,8 @@ func TestRunnerRetainsEarlierHydratedProducerWhenLaterRequiredProducerFails(t *t
loader := newAcceptedCheckpointLoader() loader := newAcceptedCheckpointLoader()
firstKey := CheckpointLaneKey(first.resolved.StepID, first.resolved.ID) firstKey := CheckpointLaneKey(first.resolved.StepID, first.resolved.ID)
secondKey := CheckpointLaneKey(second.resolved.StepID, second.resolved.ID) secondKey := CheckpointLaneKey(second.resolved.StepID, second.resolved.ID)
loader.accepted[firstKey] = NormalizeCheckpoint{Output: stored, Warnings: []contracts.Warning{{Scope: "normalize", ReasonCode: "retained-warning", Message: "retained warning"}}} stored.Diagnostics = []CheckpointDiagnostic{{Diagnostic: contracts.ProducerDiagnostic{Disposition: contracts.DiagnosticDispositionWarning, Category: contracts.DiagnosticCategoryFallback, ReasonCode: "retained-warning", OccurrenceCount: 1, Samples: []contracts.DiagnosticSample{{Scope: "normalize", Message: "retained warning"}}}}}
loader.accepted[firstKey] = NormalizeCheckpoint{Output: stored}
loader.acceptedDecision[firstKey] = NewCheckpointDecision(CheckpointDecisionReused, CheckpointReasonAcceptedArtifactReused) loader.acceptedDecision[firstKey] = NewCheckpointDecision(CheckpointDecisionReused, CheckpointReasonAcceptedArtifactReused)
loader.acceptedDecision[secondKey] = NewCheckpointDecision(CheckpointDecisionExecuted, CheckpointReasonMissing) loader.acceptedDecision[secondKey] = NewCheckpointDecision(CheckpointDecisionExecuted, CheckpointReasonMissing)
policy := CheckpointExecutionPolicy{RequireReusableLanes: map[string]struct{}{firstKey: {}, secondKey: {}}} policy := CheckpointExecutionPolicy{RequireReusableLanes: map[string]struct{}{firstKey: {}, secondKey: {}}}
@@ -240,8 +242,8 @@ func TestRunnerRetainsEarlierHydratedProducerWhenLaterRequiredProducerFails(t *t
if len(output.NormalizeOutputs) != 1 || output.NormalizeOutputs[0].LaneID != first.resolved.ID || string(output.NormalizeOutputs[0].Artifact.Content) != string(stored.Artifact.Content) { if len(output.NormalizeOutputs) != 1 || output.NormalizeOutputs[0].LaneID != first.resolved.ID || string(output.NormalizeOutputs[0].Artifact.Content) != string(stored.Artifact.Content) {
t.Fatalf("retained normalize outputs = %#v, want first producer", output.NormalizeOutputs) t.Fatalf("retained normalize outputs = %#v, want first producer", output.NormalizeOutputs)
} }
if len(output.Warnings) != 1 || output.Warnings[0].ReasonCode != "retained-warning" { if len(output.Diagnostics.Groups) != 1 || output.Diagnostics.Groups[0].ReasonCode != "retained-warning" {
t.Fatalf("retained warnings = %#v", output.Warnings) t.Fatalf("retained diagnostics = %#v", output.Diagnostics)
} }
type decisionExpectation struct { type decisionExpectation struct {
step, lane string step, lane string

View File

@@ -107,8 +107,12 @@ func (attemptDebugLLM) CompleteStructured(_ context.Context, request contracts.S
} }
func preparedAttemptDebugPipeline(t *testing.T) *PreparedPipeline { func preparedAttemptDebugPipeline(t *testing.T) *PreparedPipeline {
return preparedAttemptDebugPipelineWithChunks(t, 1)
}
func preparedAttemptDebugPipelineWithChunks(t *testing.T, chunkCount int) *PreparedPipeline {
t.Helper() t.Helper()
prepared := preparedConcurrentPipeline(t, 1) prepared := preparedConcurrentPipeline(t, chunkCount)
prepared.Steps[0].lanes = prepared.Steps[0].lanes[:1] prepared.Steps[0].lanes = prepared.Steps[0].lanes[:1]
prepared.resolved.Steps[0].ArtifactLanes = prepared.resolved.Steps[0].ArtifactLanes[:1] prepared.resolved.Steps[0].ArtifactLanes = prepared.resolved.Steps[0].ArtifactLanes[:1]
prepared.Steps[0].ArtifactLanes = prepared.Steps[0].ArtifactLanes[:1] prepared.Steps[0].ArtifactLanes = prepared.Steps[0].ArtifactLanes[:1]
@@ -162,13 +166,13 @@ func TestRunnerWritesAttemptScopedMergeAndNormalizeDebug(t *testing.T) {
if err := callAttemptDebugLLM(ctx, client, "merge"); err != nil { if err := callAttemptDebugLLM(ctx, client, "merge"); err != nil {
return erasedTypedResult{}, err return erasedTypedResult{}, err
} }
return erasedTypedResult{Value: codecNotes{Items: []string{"merged"}}, Warnings: []contracts.Warning{{Scope: "merge", ReasonCode: "observed", Message: "merge warning"}}}, nil return erasedTypedResult{Value: codecNotes{Items: []string{"merged"}}, Diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic("merge", "merge diagnostic")}}, nil
} }
prepared.Steps[0].lanes[0].typed.normalize = func(ctx context.Context, _ any, _ contracts.TypedNormalizeRequest[any]) (erasedTypedResult, error) { prepared.Steps[0].lanes[0].typed.normalize = func(ctx context.Context, _ any, _ contracts.TypedNormalizeRequest[any]) (erasedTypedResult, error) {
if err := callAttemptDebugLLM(ctx, client, "normalize"); err != nil { if err := callAttemptDebugLLM(ctx, client, "normalize"); err != nil {
return erasedTypedResult{}, err return erasedTypedResult{}, err
} }
return erasedTypedResult{Value: codecNotes{Items: []string{"normalized"}}, Warnings: []contracts.Warning{{Scope: "normalize", ReasonCode: "observed", Message: "normalize warning"}}}, nil return erasedTypedResult{Value: codecNotes{Items: []string{"normalized"}}, Diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic("normalize", "normalize diagnostic")}}, nil
} }
output, err := New().Run(context.Background(), RunInput{Prepared: prepared, RawInput: []byte("input"), Debug: debug}) output, err := New().Run(context.Background(), RunInput{Prepared: prepared, RawInput: []byte("input"), Debug: debug})
@@ -213,7 +217,7 @@ func TestRunnerWritesAttemptScopedMergeAndNormalizeDebug(t *testing.T) {
} }
} }
func TestRunnerRecordsDistinctRetryAttemptsAndPromotesAcceptedWarningsOnly(t *testing.T) { func TestRunnerRecordsDistinctRetryAttemptsAndPromotesAcceptedDiagnosticsOnly(t *testing.T) {
for _, stage := range []ModuleStage{StageMerge, StageNormalize} { for _, stage := range []ModuleStage{StageMerge, StageNormalize} {
t.Run(string(stage), func(t *testing.T) { t.Run(string(stage), func(t *testing.T) {
prepared := preparedAttemptDebugPipeline(t) prepared := preparedAttemptDebugPipeline(t)
@@ -230,7 +234,7 @@ func TestRunnerRecordsDistinctRetryAttemptsAndPromotesAcceptedWarningsOnly(t *te
if attempts == 1 { if attempts == 1 {
scope = "discarded" scope = "discarded"
} }
return erasedTypedResult{Value: codecNotes{Items: []string{scope}}, Warnings: []contracts.Warning{{Scope: scope, ReasonCode: "observed", Message: scope}}, ModelCandidate: attemptCandidate(t, fmt.Sprintf(`{"items":["%s"]}`, scope))}, nil return erasedTypedResult{Value: codecNotes{Items: []string{scope}}, Diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic(scope, scope)}, ModelCandidate: attemptCandidate(t, fmt.Sprintf(`{"items":["%s"]}`, scope))}, nil
} }
validatorCalls := 0 validatorCalls := 0
validator := preparedValidator{ validator := preparedValidator{
@@ -268,11 +272,11 @@ func TestRunnerRecordsDistinctRetryAttemptsAndPromotesAcceptedWarningsOnly(t *te
if len(first.LLMCalls) != 1 || len(second.LLMCalls) != 1 || first.LLMCalls[0].CallID == second.LLMCalls[0].CallID { if len(first.LLMCalls) != 1 || len(second.LLMCalls) != 1 || first.LLMCalls[0].CallID == second.LLMCalls[0].CallID {
t.Fatalf("retry LLM calls = first %#v, second %#v; want distinct calls", first.LLMCalls, second.LLMCalls) t.Fatalf("retry LLM calls = first %#v, second %#v; want distinct calls", first.LLMCalls, second.LLMCalls)
} }
if !strings.Contains(string(debug.json[firstPath]), "discarded") || !strings.Contains(string(debug.json[firstPath]), "rejection") { if !strings.Contains(string(debug.json[firstPath]), "rejection") {
t.Fatalf("first attempt envelope = %s, want discarded warning and rejection", debug.json[firstPath]) t.Fatalf("first attempt envelope = %s, want rejection", debug.json[firstPath])
} }
if len(output.Warnings) != 1 || output.Warnings[0].Scope != "accepted" { if len(output.Diagnostics.Groups) != 1 || output.Diagnostics.Groups[0].Samples[0].Scope != "accepted" {
t.Fatalf("promoted warnings = %#v, want accepted attempt only", output.Warnings) t.Fatalf("promoted diagnostics = %#v, want accepted attempt only", output.Diagnostics)
} }
}) })
} }

View File

@@ -101,13 +101,13 @@ type candidateCheckpointRecorder struct {
normalizeOutput CheckpointArtifact normalizeOutput CheckpointArtifact
} }
func (r *candidateCheckpointRecorder) MergeSucceeded(_ string, _ string, _ []CheckpointFingerprint, output CheckpointArtifact, _ []contracts.Warning) error { func (r *candidateCheckpointRecorder) MergeSucceeded(_ string, _ string, _ []CheckpointFingerprint, output CheckpointArtifact) error {
r.mergeSucceeded++ r.mergeSucceeded++
r.mergeOutput = cloneCheckpointArtifact(output) r.mergeOutput = cloneCheckpointArtifact(output)
return nil return nil
} }
func (r *candidateCheckpointRecorder) NormalizeSucceeded(_ string, _ string, _ []CheckpointFingerprint, output CheckpointArtifact, _ []contracts.Warning) error { func (r *candidateCheckpointRecorder) NormalizeSucceeded(_ string, _ string, _ []CheckpointFingerprint, output CheckpointArtifact) error {
r.normalizeSucceeded++ r.normalizeSucceeded++
r.normalizeOutput = cloneCheckpointArtifact(output) r.normalizeOutput = cloneCheckpointArtifact(output)
return nil return nil

View File

@@ -15,7 +15,7 @@ type chunkPlanExecution struct {
accepted bool accepted bool
chunks []source.Chunk chunks []source.Chunk
plan *source.ChunkPlan plan *source.ChunkPlan
warnings []contracts.Warning diagnostics []contracts.DiagnosticGroup
rejection *contracts.RejectedOutput rejection *contracts.RejectedOutput
lookup ChunkPlanDecision lookup ChunkPlanDecision
record *ChunkPlanRecord record *ChunkPlanRecord
@@ -28,7 +28,7 @@ type generatedChunkPlanCandidate struct {
plan source.ChunkPlan plan source.ChunkPlan
chunks []source.Chunk chunks []source.Chunk
record ChunkPlanRecord record ChunkPlanRecord
producerWarnings []contracts.Warning producerDiagnostics []contracts.ProducerDiagnostic
terminal *attemptTerminalRecorder terminal *attemptTerminalRecorder
} }
@@ -69,7 +69,7 @@ func (r *Runner) runChunkPlan(ctx context.Context, input RunInput, doc *source.S
if validationErr == nil { if validationErr == nil {
report, err := r.validateChunkReport(ctx, doc, chunker.Key(), chunks, sourceInput, sessionID, input.pipeline.ChunkReferences.ReferenceSet, input.Metadata, input.Prepared.chunkValidators, 1, input.Debug) report, err := r.validateChunkReport(ctx, doc, chunker.Key(), chunks, sourceInput, sessionID, input.pipeline.ChunkReferences.ReferenceSet, input.Metadata, input.Prepared.chunkValidators, 1, input.Debug)
if err != nil { if err != nil {
result.setValidation(report.Warnings(), nil, err) result.setValidation(nil, err)
return result, err return result, err
} }
rejection := report.FirstRejection() rejection := report.FirstRejection()
@@ -80,24 +80,28 @@ func (r *Runner) runChunkPlan(ctx context.Context, input RunInput, doc *source.S
} }
result.plan = &plan result.plan = &plan
result.chunks = chunks result.chunks = chunks
result.warnings = append(cloneWarnings(record.Warnings), report.Warnings()...)
result.accepted = true result.accepted = true
result.setValidation(report.Warnings(), nil, nil) result.setValidation(nil, nil)
cachedTerminal := producerAttemptTerminal{Action: producerTerminalAccepted, Validation: report} cachedTerminal := producerAttemptTerminal{Action: producerTerminalAccepted, Diagnostics: contracts.CloneProducerDiagnostics(record.Diagnostics), Validation: report}
diagnostics, diagnosticErr := terminalDiagnosticGroups(cachedTerminal, contracts.DiagnosticOrigin{Stage: contracts.DiagnosticOriginStageChunk, ModuleKey: chunker.Key()}, nil)
if diagnosticErr != nil {
return result, fmt.Errorf("promote reused chunk diagnostics: %w", diagnosticErr)
}
result.diagnostics = diagnostics
summary := validationSummary(cachedTerminal, StageChunk, "", "", chunker.Key(), "", 0) summary := validationSummary(cachedTerminal, StageChunk, "", "", chunker.Key(), "", 0)
result.validation = &summary result.validation = &summary
return result, nil return result, nil
} }
if incomplete != nil && input.pipeline.ChunkValidationPolicy.ValidatorFailure == ValidatorFailureFailRun { if incomplete != nil && input.pipeline.ChunkValidationPolicy.ValidatorFailure == ValidatorFailureFailRun {
failure := validatorFailureError(*incomplete) failure := validatorFailureError(*incomplete)
result.setValidation(report.Warnings(), nil, failure) result.setValidation(nil, failure)
failedTerminal := producerAttemptTerminal{Action: producerTerminalFailed, Validation: report, ValidationIncomplete: true} failedTerminal := producerAttemptTerminal{Action: producerTerminalFailed, Validation: report, ValidationIncomplete: true}
summary := validationSummary(failedTerminal, StageChunk, "", "", chunker.Key(), "", 0) summary := validationSummary(failedTerminal, StageChunk, "", "", chunker.Key(), "", 0)
result.validation = &summary result.validation = &summary
return result, failure return result, failure
} }
// A cache hit is not model material. Its rejection is discarded and // A cache hit is not model material. Its rejection is discarded and
// generation begins with the ordinary initial request below. Warnings // generation begins with the ordinary initial request below. Diagnostics
// from this discarded candidate are intentionally not promoted. // from this discarded candidate are intentionally not promoted.
} }
result.lookup = ChunkPlanDecision{Status: ChunkPlanInvalid, Reason: chunkPlanLookupReason(ChunkPlanInvalid)} result.lookup = ChunkPlanDecision{Status: ChunkPlanInvalid, Reason: chunkPlanLookupReason(ChunkPlanInvalid)}
@@ -138,7 +142,7 @@ func (r *Runner) runChunkPlan(ctx context.Context, input RunInput, doc *source.S
plan, chunks, validationErr := validateAndMaterializeChunkPlan(doc, chunkResult.Plan) plan, chunks, validationErr := validateAndMaterializeChunkPlan(doc, chunkResult.Plan)
if validationErr != nil { if validationErr != nil {
attemptErr := fmt.Errorf("validate chunk plan from chunker %q: %w", chunker.Key(), validationErr) attemptErr := fmt.Errorf("validate chunk plan from chunker %q: %w", chunker.Key(), validationErr)
payload := map[string]any{"plan": debugChunkPlanEnvelope(chunkResult.Plan), "warnings": debugWarningEnvelopes(chunkResult.Warnings)} payload := map[string]any{"plan": debugChunkPlanEnvelope(chunkResult.Plan)}
return producerAttemptOutput{}, attemptTerminal.record(payload, fmt.Errorf("%w: %v", contracts.ErrInvalidStructuredOutput, attemptErr)) return producerAttemptOutput{}, attemptTerminal.record(payload, fmt.Errorf("%w: %v", contracts.ErrInvalidStructuredOutput, attemptErr))
} }
planDigest, digestErr := source.DigestChunkPlan(plan) planDigest, digestErr := source.DigestChunkPlan(plan)
@@ -161,19 +165,18 @@ func (r *Runner) runChunkPlan(ctx context.Context, input RunInput, doc *source.S
References: append([]artifacts.ReferenceProvenance(nil), referenceTargetProvenance(input.pipeline.ChunkReferences)...), References: append([]artifacts.ReferenceProvenance(nil), referenceTargetProvenance(input.pipeline.ChunkReferences)...),
Metadata: producerMetadata, Metadata: producerMetadata,
}, },
Warnings: cloneWarnings(chunkResult.Warnings), CreatedAt: time.Now().UTC(), CreatedAt: time.Now().UTC(),
} }
return producerAttemptOutput{Value: generatedChunkPlanCandidate{plan: plan, chunks: chunks, record: candidate, producerWarnings: cloneWarnings(chunkResult.Warnings), terminal: &attemptTerminal}, Candidate: chunkResult.ModelCandidate, Warnings: cloneWarnings(chunkResult.Warnings)}, nil return producerAttemptOutput{Value: generatedChunkPlanCandidate{plan: plan, chunks: chunks, record: candidate, producerDiagnostics: contracts.CloneProducerDiagnostics(chunkResult.Diagnostics), terminal: &attemptTerminal}, Candidate: chunkResult.ModelCandidate, Diagnostics: contracts.CloneProducerDiagnostics(chunkResult.Diagnostics)}, nil
}, func(validationCtx context.Context, output producerAttemptOutput) (validationReport, error) { }, func(validationCtx context.Context, output producerAttemptOutput) (validationReport, error) {
candidate, ok := output.Value.(generatedChunkPlanCandidate) candidate, ok := output.Value.(generatedChunkPlanCandidate)
if !ok { if !ok {
return validationReport{}, fmt.Errorf("chunk attempt candidate has incompatible type") return validationReport{}, fmt.Errorf("chunk attempt candidate has incompatible type")
} }
report, validationErr := r.validateChunkReport(validationCtx, doc, chunker.Key(), candidate.chunks, sourceInput, sessionID, input.pipeline.ChunkReferences.ReferenceSet, input.Metadata, input.Prepared.chunkValidators, candidate.terminal.envelope.Attempt, input.Debug) report, validationErr := r.validateChunkReport(validationCtx, doc, chunker.Key(), candidate.chunks, sourceInput, sessionID, input.pipeline.ChunkReferences.ReferenceSet, input.Metadata, input.Prepared.chunkValidators, candidate.terminal.envelope.Attempt, input.Debug)
attemptWarnings := append(cloneWarnings(output.Warnings), report.Warnings()...)
payload := map[string]any{ payload := map[string]any{
"plan": debugChunkPlanEnvelope(candidate.plan), "materialized_chunks": debugSourceChunkEnvelopes(candidate.chunks), "plan": debugChunkPlanEnvelope(candidate.plan), "materialized_chunks": debugSourceChunkEnvelopes(candidate.chunks),
"warnings": debugWarningEnvelopes(attemptWarnings), "rejection": debugRejectedOutputPtr(chunkRejection(report, candidate.terminal.envelope.Attempt, chunker.Key())), "rejection": debugRejectedOutputPtr(chunkRejection(report, candidate.terminal.envelope.Attempt, chunker.Key())),
} }
if validationErr != nil { if validationErr != nil {
return report, candidate.terminal.record(payload, validationErr) return report, candidate.terminal.record(payload, validationErr)
@@ -190,18 +193,22 @@ func (r *Runner) runChunkPlan(ctx context.Context, input RunInput, doc *source.S
} }
return result, err return result, err
} }
diagnostics, diagnosticErr := terminalDiagnosticGroups(terminal, contracts.DiagnosticOrigin{Stage: contracts.DiagnosticOriginStageChunk, ModuleKey: chunker.Key()}, nil)
if diagnosticErr != nil {
return result, fmt.Errorf("promote chunk diagnostics: %w", diagnosticErr)
}
result.diagnostics = diagnostics
if terminal.Action == producerTerminalRejected { if terminal.Action == producerTerminalRejected {
result.rejection = terminal.Rejection result.rejection = terminal.Rejection
if result.rejection != nil { if result.rejection != nil {
result.rejection.Stage = string(StageChunk) result.rejection.Stage = string(StageChunk)
result.rejection.ModuleKey = chunker.Key() result.rejection.ModuleKey = chunker.Key()
} }
result.warnings = cloneWarnings(terminal.Warnings)
result.validation = &terminalSummary result.validation = &terminalSummary
if result.rejection != nil { if result.rejection != nil {
result.rejection.Validation = cloneValidationSummaryPtr(result.validation) result.rejection.Validation = cloneValidationSummaryPtr(result.validation)
} }
result.setValidation(terminal.Validation.Warnings(), result.rejection, nil) result.setValidation(result.rejection, nil)
return result, nil return result, nil
} }
candidate, ok := terminal.Value.(generatedChunkPlanCandidate) candidate, ok := terminal.Value.(generatedChunkPlanCandidate)
@@ -222,9 +229,8 @@ func (r *Runner) runChunkPlan(ctx context.Context, input RunInput, doc *source.S
result.accepted = true result.accepted = true
result.plan = &candidate.plan result.plan = &candidate.plan
result.chunks = candidate.chunks result.chunks = candidate.chunks
result.warnings = cloneWarnings(terminal.Warnings)
result.validation = &terminalSummary result.validation = &terminalSummary
result.setValidation(terminal.Validation.Warnings(), nil, nil) result.setValidation(nil, nil)
if terminal.ValidationIncomplete { if terminal.ValidationIncomplete {
result.summary.ValidationStatus = "incomplete" result.summary.ValidationStatus = "incomplete"
} }
@@ -234,7 +240,7 @@ func (r *Runner) runChunkPlan(ctx context.Context, input RunInput, doc *source.S
if cloneErr != nil { if cloneErr != nil {
return result, fmt.Errorf("clone chunk plan record for publication: %w", cloneErr) return result, fmt.Errorf("clone chunk plan record for publication: %w", cloneErr)
} }
record.Warnings = cloneWarnings(candidate.producerWarnings) record.Diagnostics = contracts.CloneProducerDiagnostics(candidate.producerDiagnostics)
if err := input.ChunkPlans.Save(record); err != nil { if err := input.ChunkPlans.Save(record); err != nil {
return result, fmt.Errorf("save chunk plan: %w", err) return result, fmt.Errorf("save chunk plan: %w", err)
} }
@@ -290,14 +296,12 @@ func (result *chunkPlanExecution) setCandidate(record ChunkPlanRecord, action st
return nil return nil
} }
func (result *chunkPlanExecution) setValidation(warnings []contracts.Warning, rejection *contracts.RejectedOutput, err error) { func (result *chunkPlanExecution) setValidation(rejection *contracts.RejectedOutput, err error) {
switch { switch {
case err != nil: case err != nil:
result.summary.ValidationStatus = "error" result.summary.ValidationStatus = "error"
case rejection != nil: case rejection != nil:
result.summary.ValidationStatus = "rejected" result.summary.ValidationStatus = "rejected"
case len(warnings) > 0:
result.summary.ValidationStatus = "approved_with_warnings"
default: default:
result.summary.ValidationStatus = "approved" result.summary.ValidationStatus = "approved"
} }
@@ -311,7 +315,7 @@ func cloneChunkPlanRecord(record ChunkPlanRecord) (ChunkPlanRecord, error) {
return ChunkPlanRecord{}, fmt.Errorf("clone chunk plan producer metadata: %w", err) return ChunkPlanRecord{}, fmt.Errorf("clone chunk plan producer metadata: %w", err)
} }
record.Producer.Metadata = metadata record.Producer.Metadata = metadata
record.Warnings = cloneWarnings(record.Warnings) record.Diagnostics = contracts.CloneProducerDiagnostics(record.Diagnostics)
return record, nil return record, nil
} }

View File

@@ -118,9 +118,9 @@ func (*retryingChunker) ReferenceSlots() []contracts.ReferenceSlot { return nil
func (c *retryingChunker) Plan(context.Context, contracts.ChunkRequest) (contracts.ChunkPlanResult, error) { func (c *retryingChunker) Plan(context.Context, contracts.ChunkRequest) (contracts.ChunkPlanResult, error) {
c.calls++ c.calls++
if c.calls == 1 { if c.calls == 1 {
return contracts.ChunkPlanResult{Warnings: []contracts.Warning{{Scope: "discarded", ReasonCode: "retry", Message: "discarded warning"}}}, errors.New("retry generation") return contracts.ChunkPlanResult{Diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic("discarded", "discarded diagnostic")}}, errors.New("retry generation")
} }
return contracts.ChunkPlanResult{Plan: source.CloneChunkPlan(c.plan), Warnings: []contracts.Warning{{Scope: "accepted", ReasonCode: "observed", Message: "accepted warning"}}}, nil return contracts.ChunkPlanResult{Plan: source.CloneChunkPlan(c.plan), Diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic("accepted", "accepted diagnostic")}}, nil
} }
type dependencyLoader struct { type dependencyLoader struct {
@@ -233,6 +233,37 @@ func TestRunnerChunkPlanHitUsesStoredProducerProvenance(t *testing.T) {
} }
} }
func TestRunnerReusesChunkPlanDiagnosticsWithCurrentValidatorDiagnostics(t *testing.T) {
prepared, plan := preparedTerminalDebugPipeline(t)
record := chunkPlanRecord(t, prepared, plan)
record.Diagnostics = []contracts.ProducerDiagnostic{{
Disposition: contracts.DiagnosticDispositionWarning,
Category: contracts.DiagnosticCategoryConfiguration,
ReasonCode: "stored_chunk_diagnostic",
OccurrenceCount: 1,
Samples: []contracts.DiagnosticSample{{Scope: "reference", Message: "Stored configuration signal."}},
}}
validator := &countingChunkValidator{result: contracts.ValidationResult{Approved: true, Diagnostics: []contracts.ProducerDiagnostic{{
Disposition: contracts.DiagnosticDispositionAdvisory,
Category: contracts.DiagnosticCategoryDataQuality,
ReasonCode: "current_validator_diagnostic",
OccurrenceCount: 1,
Samples: []contracts.DiagnosticSample{{Scope: "chunk", Message: "Current validator finding."}},
}}}}
prepared.chunkValidators.validators = []preparedValidator{{resolved: ResolvedValidator{Binding: Binding(validator.Name()), Target: ValidatorTargetChunk}, chunk: validator}}
output, err := New().Run(context.Background(), RunInput{Prepared: prepared, RawInput: []byte("input"), ChunkCacheMode: ChunkCacheAuto, ChunkPlans: &recordingChunkPlanStore{record: record, decision: ChunkPlanDecision{Status: ChunkPlanHit}}})
if err != nil {
t.Fatal(err)
}
if validator.calls != 1 || len(output.Diagnostics.Groups) != 2 {
t.Fatalf("validator calls = %d diagnostics = %#v", validator.calls, output.Diagnostics)
}
if output.Diagnostics.Groups[0].ReasonCode != "stored_chunk_diagnostic" || output.Diagnostics.Groups[1].ReasonCode != "current_validator_diagnostic" || output.Diagnostics.Groups[1].Origin.ValidatorKey != validator.Name() {
t.Fatalf("diagnostic groups = %#v", output.Diagnostics.Groups)
}
}
func TestRunnerProvidesAcceptedChunkMapToOutput(t *testing.T) { func TestRunnerProvidesAcceptedChunkMapToOutput(t *testing.T) {
for _, test := range []struct { for _, test := range []struct {
name string name string
@@ -368,7 +399,7 @@ func TestRunnerRegeneratesValidationIncompleteChunkPlanHit(t *testing.T) {
}}, }},
}} }}
record := chunkPlanRecord(t, prepared, plan) record := chunkPlanRecord(t, prepared, plan)
record.Warnings = []contracts.Warning{{Scope: "stored", ReasonCode: "old", Message: "discarded stored warning"}} record.Diagnostics = []contracts.ProducerDiagnostic{producerDiagnostic("stored", "discarded stored diagnostic")}
store := &recordingChunkPlanStore{record: record, decision: ChunkPlanDecision{Status: ChunkPlanHit}} store := &recordingChunkPlanStore{record: record, decision: ChunkPlanDecision{Status: ChunkPlanHit}}
debug := newCapturedDebugRecorder() debug := newCapturedDebugRecorder()
@@ -382,8 +413,8 @@ func TestRunnerRegeneratesValidationIncompleteChunkPlanHit(t *testing.T) {
if output.ChunkPlan == nil || output.ChunkPlan.Action != "generated" || output.ChunkPlan.LookupStatus != "invalid" || output.ChunkPlan.ValidationStatus != "approved" { if output.ChunkPlan == nil || output.ChunkPlan.Action != "generated" || output.ChunkPlan.LookupStatus != "invalid" || output.ChunkPlan.ValidationStatus != "approved" {
t.Fatalf("chunk plan summary = %#v", output.ChunkPlan) t.Fatalf("chunk plan summary = %#v", output.ChunkPlan)
} }
if len(output.Warnings) != 0 { if len(output.Diagnostics.Groups) != 0 {
t.Fatalf("warnings = %#v, want discarded cache warnings omitted", output.Warnings) t.Fatalf("diagnostics = %#v, want discarded cache diagnostics omitted", output.Diagnostics)
} }
assertAttemptEnvelopeSequence(t, debug, "chunk", 1) assertAttemptEnvelopeSequence(t, debug, "chunk", 1)
} }
@@ -396,7 +427,7 @@ func TestRunnerKeepsStoredPlanWhenCacheAndGeneratedValidationAreIncomplete(t *te
validator := &countingChunkValidator{err: errors.New("validator unavailable")} validator := &countingChunkValidator{err: errors.New("validator unavailable")}
prepared.chunkValidators.validators = []preparedValidator{{resolved: ResolvedValidator{Binding: Binding(validator.Name()), Target: ValidatorTargetChunk}, chunk: validator}} prepared.chunkValidators.validators = []preparedValidator{{resolved: ResolvedValidator{Binding: Binding(validator.Name()), Target: ValidatorTargetChunk}, chunk: validator}}
record := chunkPlanRecord(t, prepared, plan) record := chunkPlanRecord(t, prepared, plan)
record.Warnings = []contracts.Warning{{Scope: "stored", ReasonCode: "old", Message: "discarded stored warning"}} record.Diagnostics = []contracts.ProducerDiagnostic{producerDiagnostic("stored", "discarded stored diagnostic")}
store := &recordingChunkPlanStore{record: record, decision: ChunkPlanDecision{Status: ChunkPlanHit}} store := &recordingChunkPlanStore{record: record, decision: ChunkPlanDecision{Status: ChunkPlanHit}}
output, err := New().Run(context.Background(), RunInput{Prepared: prepared, RawInput: []byte("input"), ChunkCacheMode: ChunkCacheAuto, ChunkPlans: store}) output, err := New().Run(context.Background(), RunInput{Prepared: prepared, RawInput: []byte("input"), ChunkCacheMode: ChunkCacheAuto, ChunkPlans: store})
@@ -409,8 +440,8 @@ func TestRunnerKeepsStoredPlanWhenCacheAndGeneratedValidationAreIncomplete(t *te
if output.ChunkPlan == nil || output.ChunkPlan.Action != "generated" || output.ChunkPlan.LookupStatus != "invalid" || output.ChunkPlan.ValidationStatus != "incomplete" { if output.ChunkPlan == nil || output.ChunkPlan.Action != "generated" || output.ChunkPlan.LookupStatus != "invalid" || output.ChunkPlan.ValidationStatus != "incomplete" {
t.Fatalf("chunk plan summary = %#v", output.ChunkPlan) t.Fatalf("chunk plan summary = %#v", output.ChunkPlan)
} }
if len(output.Warnings) != 1 || output.Warnings[0].ReasonCode != "validator_execution_incomplete" { if len(output.Diagnostics.Groups) != 1 || output.Diagnostics.Groups[0].ReasonCode != "validator_execution_incomplete" {
t.Fatalf("warnings = %#v, want only generated incomplete warning", output.Warnings) t.Fatalf("diagnostics = %#v, want only generated incomplete diagnostic", output.Diagnostics)
} }
if !reflect.DeepEqual(store.record, record) { if !reflect.DeepEqual(store.record, record) {
t.Fatal("discarded incomplete candidates mutated the stored record") t.Fatal("discarded incomplete candidates mutated the stored record")
@@ -580,8 +611,8 @@ func TestRunnerRetriesBeforePublishingAcceptedPlan(t *testing.T) {
if chunker.calls != 2 || store.saves != 1 { if chunker.calls != 2 || store.saves != 1 {
t.Fatalf("module calls = %d saves = %d, want 2 and 1", chunker.calls, store.saves) t.Fatalf("module calls = %d saves = %d, want 2 and 1", chunker.calls, store.saves)
} }
if len(output.Warnings) != 1 || output.Warnings[0].Scope != "accepted" || len(store.saved.Warnings) != 1 || store.saved.Warnings[0].Scope != "accepted" { if len(output.Diagnostics.Groups) != 1 || output.Diagnostics.Groups[0].Samples[0].Scope != "accepted" || len(store.saved.Diagnostics) != 1 || store.saved.Diagnostics[0].Samples[0].Scope != "accepted" {
t.Fatalf("output warnings = %#v stored warnings = %#v", output.Warnings, store.saved.Warnings) t.Fatalf("output diagnostics = %#v stored diagnostics = %#v", output.Diagnostics, store.saved.Diagnostics)
} }
} }
@@ -593,7 +624,7 @@ func TestRunnerAutoHitValidatesOnceWithoutRegenerationOrMutation(t *testing.T) {
wantError string wantError string
wantReject bool wantReject bool
}{ }{
{name: "warning", result: contracts.ValidationResult{Approved: true, Warnings: []contracts.Warning{{Scope: "current", ReasonCode: "observed", Message: "current warning"}}}}, {name: "diagnostic", result: contracts.ValidationResult{Approved: true, Diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic("current", "current diagnostic")}}},
{name: "rejection", result: contracts.ValidationResult{Approved: false, ReasonCode: "rejected", Message: "rejected hit", CorrectionGuidance: "return an acceptable chunk plan"}, wantReject: true}, {name: "rejection", result: contracts.ValidationResult{Approved: false, ReasonCode: "rejected", Message: "rejected hit", CorrectionGuidance: "return an acceptable chunk plan"}, wantReject: true},
{name: "error", validatorErr: errors.New("validator failed"), wantError: "validator failed"}, {name: "error", validatorErr: errors.New("validator failed"), wantError: "validator failed"},
} }
@@ -606,7 +637,7 @@ func TestRunnerAutoHitValidatesOnceWithoutRegenerationOrMutation(t *testing.T) {
validator := &countingChunkValidator{result: tc.result, err: tc.validatorErr} validator := &countingChunkValidator{result: tc.result, err: tc.validatorErr}
prepared.chunkValidators.validators = []preparedValidator{{resolved: ResolvedValidator{Binding: Binding(validator.Name()), Target: ValidatorTargetChunk}, chunk: validator}} prepared.chunkValidators.validators = []preparedValidator{{resolved: ResolvedValidator{Binding: Binding(validator.Name()), Target: ValidatorTargetChunk}, chunk: validator}}
record := chunkPlanRecord(t, prepared, plan) record := chunkPlanRecord(t, prepared, plan)
record.Warnings = []contracts.Warning{{Scope: "stored", ReasonCode: "observed", Message: "stored warning"}} record.Diagnostics = []contracts.ProducerDiagnostic{producerDiagnostic("stored", "stored diagnostic")}
store := &recordingChunkPlanStore{record: record, decision: ChunkPlanDecision{Status: ChunkPlanHit}} store := &recordingChunkPlanStore{record: record, decision: ChunkPlanDecision{Status: ChunkPlanHit}}
debug := newCapturedDebugRecorder() debug := newCapturedDebugRecorder()
output, err := New().Run(context.Background(), RunInput{Prepared: prepared, RawInput: []byte("input"), ChunkCacheMode: ChunkCacheAuto, ChunkPlans: store, Debug: debug}) output, err := New().Run(context.Background(), RunInput{Prepared: prepared, RawInput: []byte("input"), ChunkCacheMode: ChunkCacheAuto, ChunkPlans: store, Debug: debug})
@@ -642,8 +673,8 @@ func TestRunnerAutoHitValidatesOnceWithoutRegenerationOrMutation(t *testing.T) {
if tc.wantReject && (len(output.Rejected) != 1 || output.Rejected[0].ReasonCode != "rejected") { if tc.wantReject && (len(output.Rejected) != 1 || output.Rejected[0].ReasonCode != "rejected") {
t.Fatalf("rejected = %#v", output.Rejected) t.Fatalf("rejected = %#v", output.Rejected)
} }
if tc.wantError == "" && !tc.wantReject && len(output.Warnings) != 1+len(tc.result.Warnings) { if tc.wantError == "" && !tc.wantReject && len(output.Diagnostics.Groups) != 1+len(tc.result.Diagnostics) {
t.Fatalf("warnings = %#v, want stored warning once plus current warnings", output.Warnings) t.Fatalf("diagnostics = %#v, want stored diagnostic once plus current diagnostics", output.Diagnostics)
} }
}) })
} }
@@ -693,7 +724,7 @@ func TestRunnerPublishesOnlyAcceptedGeneratedPlans(t *testing.T) {
} }
} }
func TestRunnerStoresProducerProvenanceAndProducerWarnings(t *testing.T) { func TestRunnerStoresProducerProvenanceAndDiagnostics(t *testing.T) {
prepared, plan := preparedTerminalDebugPipeline(t) prepared, plan := preparedTerminalDebugPipeline(t)
prepared.resolved.Chunk.LLMProfile = "chunk-profile" prepared.resolved.Chunk.LLMProfile = "chunk-profile"
prepared.resolved.ChunkExecutionClass = contracts.ExecutionClassLLMBacked prepared.resolved.ChunkExecutionClass = contracts.ExecutionClassLLMBacked
@@ -703,8 +734,8 @@ func TestRunnerStoresProducerProvenanceAndProducerWarnings(t *testing.T) {
"guide": {Items: []contracts.ReferenceItem{{SlotName: "guide", Digest: "sha256:guide", Origin: contracts.ReferenceOrigin{Type: "file", URI: "file:///guide.txt"}, Content: []byte("sensitive")}}}, "guide": {Items: []contracts.ReferenceItem{{SlotName: "guide", Digest: "sha256:guide", Origin: contracts.ReferenceOrigin{Type: "file", URI: "file:///guide.txt"}, Content: []byte("sensitive")}}},
}}, }},
} }
prepared.chunker = manifestChunker{terminalChunker: terminalChunker{key: prepared.resolved.Chunk.Module, plan: plan, warnings: []contracts.Warning{{Scope: "producer", ReasonCode: "observed", Message: "producer warning"}}}, metadata: map[string]any{"prompt_id": "chunk/prompt"}} prepared.chunker = manifestChunker{terminalChunker: terminalChunker{key: prepared.resolved.Chunk.Module, plan: plan, diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic("producer", "producer diagnostic")}}, metadata: map[string]any{"prompt_id": "chunk/prompt"}}
validator := &countingChunkValidator{result: contracts.ValidationResult{Approved: true, Warnings: []contracts.Warning{{Scope: "validator", ReasonCode: "observed", Message: "validator warning"}}}} validator := &countingChunkValidator{result: contracts.ValidationResult{Approved: true, Diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic("validator", "validator diagnostic")}}}
prepared.chunkValidators.validators = []preparedValidator{{resolved: ResolvedValidator{Binding: Binding(validator.Name()), Target: ValidatorTargetChunk}, chunk: validator}} prepared.chunkValidators.validators = []preparedValidator{{resolved: ResolvedValidator{Binding: Binding(validator.Name()), Target: ValidatorTargetChunk}, chunk: validator}}
store := &recordingChunkPlanStore{} store := &recordingChunkPlanStore{}
if _, err := New().Run(context.Background(), RunInput{Prepared: prepared, RawInput: []byte("input"), ChunkCacheMode: ChunkCacheRefresh, ChunkPlans: store}); err != nil { if _, err := New().Run(context.Background(), RunInput{Prepared: prepared, RawInput: []byte("input"), ChunkCacheMode: ChunkCacheRefresh, ChunkPlans: store}); err != nil {
@@ -717,8 +748,8 @@ func TestRunnerStoresProducerProvenanceAndProducerWarnings(t *testing.T) {
if len(producer.References) != 1 || producer.References[0].Digest != "sha256:guide" { if len(producer.References) != 1 || producer.References[0].Digest != "sha256:guide" {
t.Fatalf("producer references = %#v", producer.References) t.Fatalf("producer references = %#v", producer.References)
} }
if len(store.saved.Warnings) != 1 || store.saved.Warnings[0].Scope != "producer" { if len(store.saved.Diagnostics) != 1 || store.saved.Diagnostics[0].Samples[0].Scope != "producer" {
t.Fatalf("stored warnings = %#v, want only producer warning", store.saved.Warnings) t.Fatalf("stored diagnostics = %#v, want only producer diagnostic", store.saved.Diagnostics)
} }
} }

View File

@@ -147,7 +147,6 @@ func (e *cancelingOutputEncoder) Encode(context.Context, contracts.OutputRequest
e.cancel() e.cancel()
return contracts.OutputResult{ return contracts.OutputResult{
Files: []contracts.OutputFile{{Name: "result.txt", ContentType: "text/plain", Bytes: []byte("result")}}, Files: []contracts.OutputFile{{Name: "result.txt", ContentType: "text/plain", Bytes: []byte("result")}},
Warnings: []contracts.Warning{{ReasonCode: "returned-after-cancel"}},
}, nil }, nil
} }
@@ -202,9 +201,9 @@ func TestRunnerDiscardsOutputReturnedAfterCancellation(t *testing.T) {
if len(output.OutputFiles) != 0 { if len(output.OutputFiles) != 0 {
t.Fatalf("output files = %#v, want none", output.OutputFiles) t.Fatalf("output files = %#v, want none", output.OutputFiles)
} }
for _, warning := range output.Warnings { for _, group := range output.Diagnostics.Groups {
if warning.ReasonCode == "returned-after-cancel" { if group.ReasonCode == "returned-after-cancel" {
t.Fatalf("output warnings include encoder warning after cancellation") t.Fatalf("output diagnostics include encoder diagnostic after cancellation")
} }
} }
} }
@@ -367,7 +366,8 @@ func TestRunnerBoundsExtractJobsAndStabilizesReverseCompletion(t *testing.T) {
case <-ctx.Done(): case <-ctx.Done():
return erasedTypedResult{}, ctx.Err() return erasedTypedResult{}, ctx.Err()
} }
return erasedTypedResult{Value: typedValueForLane(lane, request.Chunk.Index), Warnings: []contracts.Warning{{Scope: fmt.Sprintf("lane-%d/chunk-%d", lane, request.Chunk.Index), ReasonCode: "observed", Message: "ordered"}}}, nil scope := fmt.Sprintf("lane-%d/chunk-%d", lane, request.Chunk.Index)
return erasedTypedResult{Value: typedValueForLane(lane, request.Chunk.Index), Diagnostics: []contracts.ProducerDiagnostic{{Disposition: contracts.DiagnosticDispositionObservation, Category: contracts.DiagnosticCategoryNormalization, ReasonCode: "observed", OccurrenceCount: 1, Samples: []contracts.DiagnosticSample{{Scope: scope, Message: "ordered"}}}}}, nil
}) })
} }
@@ -396,13 +396,23 @@ func TestRunnerBoundsExtractJobsAndStabilizesReverseCompletion(t *testing.T) {
if got := maximum.Load(); got != 2 { if got := maximum.Load(); got != 2 {
t.Fatalf("maximum concurrent extract jobs = %d, want 2", got) t.Fatalf("maximum concurrent extract jobs = %d, want 2", got)
} }
wantScopes := []string{"lane-0/chunk-0", "lane-0/chunk-1", "lane-0/chunk-2", "lane-1/chunk-0", "lane-1/chunk-1", "lane-1/chunk-2"} if len(result.output.Diagnostics.Groups) != 2 {
gotScopes := make([]string, len(result.output.Warnings)) t.Fatalf("diagnostic groups = %#v, want one deterministic group per lane", result.output.Diagnostics.Groups)
for i := range result.output.Warnings { }
gotScopes[i] = result.output.Warnings[i].Scope for lane, group := range result.output.Diagnostics.Groups {
if group.Origin.LaneID != prepared.Steps[0].lanes[lane].resolved.ID {
t.Fatalf("group origin = %#v, want configured lane %q", group.Origin, prepared.Steps[0].lanes[lane].resolved.ID)
}
indexes := make([]int, len(group.Samples))
for index, sample := range group.Samples {
if sample.ChunkIndex == nil {
t.Fatalf("sample = %#v, want chunk index", sample)
}
indexes[index] = *sample.ChunkIndex
}
if !reflect.DeepEqual(indexes, []int{0, 1, 2}) {
t.Fatalf("group sample chunk order = %#v, want canonical chunk order", indexes)
} }
if !reflect.DeepEqual(gotScopes, wantScopes) {
t.Fatalf("warning order = %#v, want %#v", gotScopes, wantScopes)
} }
} }

View File

@@ -24,7 +24,7 @@ type laneExtractState struct {
reuseEligible bool reuseEligible bool
values []erasedExtractArtifact values []erasedExtractArtifact
serialized []CheckpointArtifact serialized []CheckpointArtifact
warnings []contracts.Warning diagnostics []contracts.DiagnosticGroup
rejected []contracts.RejectedOutput rejected []contracts.RejectedOutput
incomplete []int incomplete []int
validationSummaries []artifacts.ValidationSummary validationSummaries []artifacts.ValidationSummary
@@ -38,7 +38,7 @@ type laneExtractState struct {
type finalizedExtractResults struct { type finalizedExtractResults struct {
accepted []erasedExtractArtifact accepted []erasedExtractArtifact
serialized []CheckpointArtifact serialized []CheckpointArtifact
warnings []contracts.Warning diagnostics []contracts.DiagnosticGroup
rejected []contracts.RejectedOutput rejected []contracts.RejectedOutput
incomplete []int incomplete []int
validationSummaries []artifacts.ValidationSummary validationSummaries []artifacts.ValidationSummary
@@ -53,8 +53,8 @@ func loadExtract(loader CheckpointLoader, stepID, laneID, moduleKey string, deps
return loader.Extract(laneID, moduleKey, deps) return loader.Extract(laneID, moduleKey, deps)
} }
func recordExtract(recorder CheckpointRecorder, stepID, laneID, moduleKey string, deps []CheckpointFingerprint, outputs []CheckpointArtifact, rejected []contracts.RejectedOutput, warnings []contracts.Warning) error { func recordExtract(recorder CheckpointRecorder, stepID, laneID, moduleKey string, deps []CheckpointFingerprint, outputs []CheckpointArtifact, rejected []contracts.RejectedOutput) error {
return checkpointExtractSucceeded(recorder, stepID, laneID, moduleKey, deps, outputs, rejected, warnings) return checkpointExtractSucceeded(recorder, stepID, laneID, moduleKey, deps, outputs, rejected)
} }
type extractJob struct { type extractJob struct {
@@ -67,7 +67,7 @@ type extractJobResult struct {
chunkIndex int chunkIndex int
value erasedExtractArtifact value erasedExtractArtifact
serialized CheckpointArtifact serialized CheckpointArtifact
warnings []contracts.Warning diagnostics []contracts.DiagnosticGroup
rejected *contracts.RejectedOutput rejected *contracts.RejectedOutput
validationIncomplete bool validationIncomplete bool
validationSummary *artifacts.ValidationSummary validationSummary *artifacts.ValidationSummary
@@ -353,7 +353,11 @@ func hydrateRequiredLane(input RunInput, loader CheckpointLoader, doc *source.So
return state, err return state, err
} }
hydrated := resolution.artifacts[0] hydrated := resolution.artifacts[0]
local.Warnings = append(local.Warnings, cloneWarnings(checkpoint.Warnings)...) diagnostics, diagnosticErr := promoteCheckpointDiagnostics(hydrated.Diagnostics, contracts.DiagnosticOrigin{Stage: contracts.DiagnosticOriginStageNormalize, StepID: input.stepID, LaneID: lane.ID, ModuleKey: lane.Normalize.Module}, nil)
if diagnosticErr != nil {
return state, fmt.Errorf("promote reused accepted diagnostics: %w", diagnosticErr)
}
appendDiagnosticGroups(&local, diagnostics)
local.NormalizeOutputs = append(local.NormalizeOutputs, contracts.SerializedOutput{ local.NormalizeOutputs = append(local.NormalizeOutputs, contracts.SerializedOutput{
StepID: input.stepID, StepID: input.stepID,
LaneID: lane.ID, LaneID: lane.ID,
@@ -418,8 +422,14 @@ func prepareLaneExtract(input RunInput, loader CheckpointLoader, doc *source.Sou
} }
state.values = append(state.values, artifact) state.values = append(state.values, artifact)
state.serialized = append(state.serialized, cloneCheckpointArtifact(stored)) state.serialized = append(state.serialized, cloneCheckpointArtifact(stored))
chunk := source.Chunk{ID: stored.ChunkID, Index: stored.ChunkIndex}
diagnostics, diagnosticErr := promoteCheckpointDiagnostics(stored.Diagnostics, contracts.DiagnosticOrigin{Stage: contracts.DiagnosticOriginStageExtract, StepID: input.stepID, LaneID: lane.ID, ModuleKey: lane.Extract.Module}, &chunk)
if diagnosticErr != nil {
return nil, fmt.Errorf("promote reused extract diagnostics: %w", diagnosticErr)
} }
state.warnings, state.rejected = cloneWarnings(cp.Warnings), cloneRejectedOutputs(cp.Rejected) state.diagnostics = append(state.diagnostics, diagnostics...)
}
state.rejected = cloneRejectedOutputs(cp.Rejected)
} }
return state, nil return state, nil
} }
@@ -450,15 +460,14 @@ func (r *Runner) runExtractJob(ctx context.Context, input RunInput, doc *source.
return producerAttemptOutput{}, terminal.record(nil, attemptErr) return producerAttemptOutput{}, terminal.record(nil, attemptErr)
} }
artifact := erasedExtractArtifact{LaneID: lane.ID, ExtractorKey: lane.Extract.Module, SourceID: doc.ID, ChunkID: chunk.ID, ChunkIndex: chunk.Index, ChunkRef: chunk.Ref, Value: extracted.Value} artifact := erasedExtractArtifact{LaneID: lane.ID, ExtractorKey: lane.Extract.Module, SourceID: doc.ID, ChunkID: chunk.ID, ChunkIndex: chunk.Index, ChunkRef: chunk.Ref, Value: extracted.Value}
attemptWarnings := cloneWarnings(extracted.Warnings)
serializedCandidate, encodeErr := serializeCandidateArtifact(typed.codec, artifact.LaneID, artifact.ExtractorKey, artifact.SourceID, artifact.Value) serializedCandidate, encodeErr := serializeCandidateArtifact(typed.codec, artifact.LaneID, artifact.ExtractorKey, artifact.SourceID, artifact.Value)
if encodeErr != nil { if encodeErr != nil {
attemptErr := fmt.Errorf("serialize extract candidate for lane %q chunk %q: %w", lane.ID, chunk.ID, encodeErr) attemptErr := fmt.Errorf("serialize extract candidate for lane %q chunk %q: %w", lane.ID, chunk.ID, encodeErr)
payload := map[string]any{"warnings": debugWarningEnvelopes(attemptWarnings)} payload := map[string]any{}
return producerAttemptOutput{}, terminal.record(payload, attemptErr) return producerAttemptOutput{}, terminal.record(payload, attemptErr)
} }
serializedCandidate.ChunkID, serializedCandidate.ChunkIndex, serializedCandidate.ChunkRef = artifact.ChunkID, artifact.ChunkIndex, artifact.ChunkRef serializedCandidate.ChunkID, serializedCandidate.ChunkIndex, serializedCandidate.ChunkRef = artifact.ChunkID, artifact.ChunkIndex, artifact.ChunkRef
return producerAttemptOutput{Value: extractAttemptValue{artifact: artifact, serialized: serializedCandidate, terminal: &terminal}, Candidate: extracted.ModelCandidate, Warnings: attemptWarnings}, nil return producerAttemptOutput{Value: extractAttemptValue{artifact: artifact, serialized: serializedCandidate, terminal: &terminal}, Candidate: extracted.ModelCandidate, Diagnostics: contracts.CloneProducerDiagnostics(extracted.Diagnostics)}, nil
}, func(validationCtx context.Context, output producerAttemptOutput) (validationReport, error) { }, func(validationCtx context.Context, output producerAttemptOutput) (validationReport, error) {
candidate, ok := output.Value.(extractAttemptValue) candidate, ok := output.Value.(extractAttemptValue)
if !ok { if !ok {
@@ -467,7 +476,6 @@ func (r *Runner) runExtractJob(ctx context.Context, input RunInput, doc *source.
report, validationErr := r.validateTypedReport(validationCtx, typed.codec, typedValidationTarget{stage: StageExtract, stepID: input.stepID, laneID: lane.ID, moduleKey: lane.Extract.Module, source: doc, sourceID: doc.ID, sourceInput: chunkInputMaterial(sourceInput, chunk), sessionID: sessionID, references: operationReferenceSet(input, lane.ExtractReferences), metadata: input.Metadata, chunk: &chunk, ref: chunk.Ref, value: candidate.artifact.Value, candidate: &candidate.serialized}, state.prepared.extractValidators, candidate.terminal.envelope.Attempt, input.Debug) report, validationErr := r.validateTypedReport(validationCtx, typed.codec, typedValidationTarget{stage: StageExtract, stepID: input.stepID, laneID: lane.ID, moduleKey: lane.Extract.Module, source: doc, sourceID: doc.ID, sourceInput: chunkInputMaterial(sourceInput, chunk), sessionID: sessionID, references: operationReferenceSet(input, lane.ExtractReferences), metadata: input.Metadata, chunk: &chunk, ref: chunk.Ref, value: candidate.artifact.Value, candidate: &candidate.serialized}, state.prepared.extractValidators, candidate.terminal.envelope.Attempt, input.Debug)
payload := map[string]any{ payload := map[string]any{
"output": debugCheckpointArtifact(candidate.serialized), "output": debugCheckpointArtifact(candidate.serialized),
"warnings": debugWarningEnvelopes(append(cloneWarnings(output.Warnings), report.Warnings()...)),
"rejection": debugRejectedOutputPtr(typedRejection(report, typedValidationTarget{stage: StageExtract, stepID: input.stepID, laneID: lane.ID, moduleKey: lane.Extract.Module, chunk: &chunk}, candidate.terminal.envelope.Attempt)), "rejection": debugRejectedOutputPtr(typedRejection(report, typedValidationTarget{stage: StageExtract, stepID: input.stepID, laneID: lane.ID, moduleKey: lane.Extract.Module, chunk: &chunk}, candidate.terminal.envelope.Attempt)),
} }
if validationErr != nil { if validationErr != nil {
@@ -481,6 +489,14 @@ func (r *Runner) runExtractJob(ctx context.Context, input RunInput, doc *source.
result.err = errors.Join(result.err, debugErr) result.err = errors.Join(result.err, debugErr)
return result return result
} }
if err == nil {
diagnostics, diagnosticErr := terminalDiagnosticGroups(terminalResult, contracts.DiagnosticOrigin{Stage: contracts.DiagnosticOriginStageExtract, StepID: input.stepID, LaneID: lane.ID, ModuleKey: lane.Extract.Module}, &chunk)
if diagnosticErr != nil {
result.err = fmt.Errorf("promote extract diagnostics: %w", diagnosticErr)
return result
}
result.diagnostics = diagnostics
}
if err == nil && terminalResult.Action == producerTerminalRejected { if err == nil && terminalResult.Action == producerTerminalRejected {
result.rejected = terminalResult.Rejection result.rejected = terminalResult.Rejection
if result.rejected != nil { if result.rejected != nil {
@@ -488,7 +504,6 @@ func (r *Runner) runExtractJob(ctx context.Context, input RunInput, doc *source.
result.validationSummary = &summary result.validationSummary = &summary
result.rejected.Validation = cloneValidationSummaryPtr(result.validationSummary) result.rejected.Validation = cloneValidationSummaryPtr(result.validationSummary)
} }
result.warnings = cloneWarnings(terminalResult.Warnings)
return result return result
} }
if err == nil { if err == nil {
@@ -498,19 +513,19 @@ func (r *Runner) runExtractJob(ctx context.Context, input RunInput, doc *source.
return result return result
} }
stored, encodeErr := checkpointArtifact(typed.codec, candidate.artifact.LaneID, candidate.artifact.ExtractorKey, candidate.artifact.SourceID, candidate.artifact.Value) stored, encodeErr := checkpointArtifact(typed.codec, candidate.artifact.LaneID, candidate.artifact.ExtractorKey, candidate.artifact.SourceID, candidate.artifact.Value)
payload := map[string]any{"output": debugCheckpointArtifact(candidate.serialized), "warnings": debugWarningEnvelopes(terminalResult.Warnings), "rejection": debugRejectedOutputPtr(nil)} payload := map[string]any{"output": debugCheckpointArtifact(candidate.serialized), "rejection": debugRejectedOutputPtr(nil)}
if encodeErr != nil { if encodeErr != nil {
attemptErr := fmt.Errorf("serialize accepted extract output for lane %q chunk %q: %w", lane.ID, chunk.ID, encodeErr) attemptErr := fmt.Errorf("serialize accepted extract output for lane %q chunk %q: %w", lane.ID, chunk.ID, encodeErr)
result.err = candidate.terminal.record(payload, attemptErr) result.err = candidate.terminal.record(payload, attemptErr)
return result return result
} }
stored.ChunkID, stored.ChunkIndex, stored.ChunkRef = candidate.artifact.ChunkID, candidate.artifact.ChunkIndex, candidate.artifact.ChunkRef stored.ChunkID, stored.ChunkIndex, stored.ChunkRef = candidate.artifact.ChunkID, candidate.artifact.ChunkIndex, candidate.artifact.ChunkRef
stored.Diagnostics = terminalCheckpointDiagnostics(terminalResult)
if debugErr := candidate.terminal.record(payload, nil); debugErr != nil { if debugErr := candidate.terminal.record(payload, nil); debugErr != nil {
result.err = debugErr result.err = debugErr
return result return result
} }
result.value, result.serialized = candidate.artifact, stored result.value, result.serialized = candidate.artifact, stored
result.warnings = cloneWarnings(terminalResult.Warnings)
result.validationIncomplete = terminalResult.ValidationIncomplete result.validationIncomplete = terminalResult.ValidationIncomplete
result.validationSummary = &summary result.validationSummary = &summary
} }
@@ -531,12 +546,12 @@ func finalizeLaneExtract(checkpoints CheckpointRecorder, stepID string, state *l
} }
if result.rejected != nil { if result.rejected != nil {
state.rejected = append(state.rejected, *result.rejected) state.rejected = append(state.rejected, *result.rejected)
state.warnings = append(state.warnings, result.warnings...) state.diagnostics = append(state.diagnostics, result.diagnostics...)
continue continue
} }
state.values = append(state.values, result.value) state.values = append(state.values, result.value)
state.serialized = append(state.serialized, result.serialized) state.serialized = append(state.serialized, result.serialized)
state.warnings = append(state.warnings, result.warnings...) state.diagnostics = append(state.diagnostics, result.diagnostics...)
if result.validationIncomplete { if result.validationIncomplete {
state.incomplete = append(state.incomplete, result.chunkIndex) state.incomplete = append(state.incomplete, result.chunkIndex)
} }
@@ -549,7 +564,7 @@ func finalizeLaneExtract(checkpoints CheckpointRecorder, stepID string, state *l
state.reuseEligible = false state.reuseEligible = false
} }
if !state.decision.Reused && state.reuseEligible { if !state.decision.Reused && state.reuseEligible {
if err := recordExtract(checkpoints, stepID, lane.ID, lane.Extract.Module, state.deps, state.serialized, state.rejected, state.warnings); err != nil { if err := recordExtract(checkpoints, stepID, lane.ID, lane.Extract.Module, state.deps, state.serialized, state.rejected); err != nil {
return fmt.Errorf("write extract checkpoint for lane %q: %w", lane.ID, err) return fmt.Errorf("write extract checkpoint for lane %q: %w", lane.ID, err)
} }
} }
@@ -562,20 +577,20 @@ func (r *Runner) continueLane(ctx context.Context, input RunInput, checkpoints C
results := finalizedExtractResults{ results := finalizedExtractResults{
accepted: state.values, accepted: state.values,
serialized: state.serialized, serialized: state.serialized,
warnings: state.warnings, diagnostics: state.diagnostics,
rejected: state.rejected, rejected: state.rejected,
incomplete: state.incomplete, incomplete: state.incomplete,
validationSummaries: state.validationSummaries, validationSummaries: state.validationSummaries,
decision: state.decision, decision: state.decision,
reuseEligible: state.reuseEligible, reuseEligible: state.reuseEligible,
} }
local.Warnings = append(local.Warnings, cloneWarnings(results.warnings)...) local.diagnosticGroups = append(local.diagnosticGroups, contracts.CloneDiagnosticCollection(contracts.DiagnosticCollection{Groups: results.diagnostics}).Groups...)
local.Rejected = append(local.Rejected, cloneRejectedOutputs(results.rejected)...) local.Rejected = append(local.Rejected, cloneRejectedOutputs(results.rejected)...)
local.ValidationSummaries = append(local.ValidationSummaries, cloneValidationSummaries(results.validationSummaries)...) local.ValidationSummaries = append(local.ValidationSummaries, cloneValidationSummaries(results.validationSummaries)...)
if err := writeDebugTimed(input.Debug, path.Join("extract", fileio.EncodePathComponent(lane.ID), "input.json"), debugTimedEnvelope{Stage: string(StageExtract), StepID: input.stepID, LaneID: lane.ID, ModuleKey: lane.Extract.Module, StartedAt: time.Now().UTC(), Payload: map[string]any{"reused": results.decision.Reused, "decision": results.decision, "source": debugSourceDocumentEnvelope(doc), "chunks": debugSourceChunkEnvelopes(chunks), "options": redactSensitiveMap(lane.Extract.Options), "metadata": redactSensitiveMap(input.Metadata)}}); err != nil { if err := writeDebugTimed(input.Debug, path.Join("extract", fileio.EncodePathComponent(lane.ID), "input.json"), debugTimedEnvelope{Stage: string(StageExtract), StepID: input.stepID, LaneID: lane.ID, ModuleKey: lane.Extract.Module, StartedAt: time.Now().UTC(), Payload: map[string]any{"reused": results.decision.Reused, "decision": results.decision, "source": debugSourceDocumentEnvelope(doc), "chunks": debugSourceChunkEnvelopes(chunks), "options": redactSensitiveMap(lane.Extract.Options), "metadata": redactSensitiveMap(input.Metadata)}}); err != nil {
return local, &laneRunError{stage: StageExtract, err: err} return local, &laneRunError{stage: StageExtract, err: err}
} }
if err := writeDebugTimed(input.Debug, path.Join("extract", fileio.EncodePathComponent(lane.ID), "output.json"), debugTimedEnvelope{Stage: string(StageExtract), StepID: input.stepID, LaneID: lane.ID, ModuleKey: lane.Extract.Module, StartedAt: time.Now().UTC(), Payload: map[string]any{"reused": results.decision.Reused, "outputs": debugCheckpointArtifacts(results.serialized), "rejected": debugRejectedOutputEnvelopes(results.rejected), "warnings": debugWarningEnvelopes(results.warnings), "validation_incomplete_chunks": append([]int(nil), results.incomplete...)}}); err != nil { if err := writeDebugTimed(input.Debug, path.Join("extract", fileio.EncodePathComponent(lane.ID), "output.json"), debugTimedEnvelope{Stage: string(StageExtract), StepID: input.stepID, LaneID: lane.ID, ModuleKey: lane.Extract.Module, StartedAt: time.Now().UTC(), Payload: map[string]any{"reused": results.decision.Reused, "outputs": debugCheckpointArtifacts(results.serialized), "rejected": debugRejectedOutputEnvelopes(results.rejected), "validation_incomplete_chunks": append([]int(nil), results.incomplete...)}}); err != nil {
return local, &laneRunError{stage: StageExtract, err: err} return local, &laneRunError{stage: StageExtract, err: err}
} }
if len(results.accepted) == 0 { if len(results.accepted) == 0 {
@@ -640,7 +655,7 @@ func mergeLaneOutput(dst *RunOutput, src RunOutput) error {
} }
dst.NormalizeOutputs = append(dst.NormalizeOutputs, cloneSerializedOutputs(src.NormalizeOutputs)...) dst.NormalizeOutputs = append(dst.NormalizeOutputs, cloneSerializedOutputs(src.NormalizeOutputs)...)
dst.Rejected = append(dst.Rejected, cloneRejectedOutputs(src.Rejected)...) dst.Rejected = append(dst.Rejected, cloneRejectedOutputs(src.Rejected)...)
dst.Warnings = append(dst.Warnings, cloneWarnings(src.Warnings)...) appendDiagnosticGroups(dst, src.diagnosticGroups)
dst.CheckpointEvents = append(dst.CheckpointEvents, src.CheckpointEvents...) dst.CheckpointEvents = append(dst.CheckpointEvents, src.CheckpointEvents...)
dst.ValidationSummaries = append(dst.ValidationSummaries, cloneValidationSummaries(src.ValidationSummaries)...) dst.ValidationSummaries = append(dst.ValidationSummaries, cloneValidationSummaries(src.ValidationSummaries)...)
if len(src.normalizeReuseEligibility) > 0 { if len(src.normalizeReuseEligibility) > 0 {

View File

@@ -4,7 +4,6 @@ import (
"context" "context"
"encoding/json" "encoding/json"
"fmt" "fmt"
"reflect"
"strings" "strings"
"sync" "sync"
"testing" "testing"
@@ -85,15 +84,6 @@ func TestRunnerCorrectsExtractCandidatesIndependentlyPerChunk(t *testing.T) {
} }
} }
func containsWarnings(have, want []contracts.Warning) bool {
for index := 0; index+len(want) <= len(have); index++ {
if reflect.DeepEqual(have[index:index+len(want)], want) {
return true
}
}
return false
}
func containsValidationSummary(summaries []artifacts.ValidationSummary, status string) bool { func containsValidationSummary(summaries []artifacts.ValidationSummary, status string) bool {
for _, summary := range summaries { for _, summary := range summaries {
if summary.Status == status { if summary.Status == status {
@@ -126,8 +116,8 @@ func TestRunnerContinuesValidationIncompleteExtractWithoutCheckpoint(t *testing.
if output.Manifest.ValidationStatus != "incomplete" || !containsValidationSummary(output.Manifest.ValidationSummaries, "incomplete") { if output.Manifest.ValidationStatus != "incomplete" || !containsValidationSummary(output.Manifest.ValidationSummaries, "incomplete") {
t.Fatalf("manifest validation = %#v, want incomplete extract provenance", output.Manifest) t.Fatalf("manifest validation = %#v, want incomplete extract provenance", output.Manifest)
} }
if got, want := output.Warnings, []contracts.Warning{{Scope: "typed/check", ReasonCode: "validator_execution_incomplete", Message: "Validator execution did not complete within its configured budget."}}; !containsWarnings(got, want) { if len(output.Diagnostics.Groups) != 1 || output.Diagnostics.Groups[0].ReasonCode != "validator_execution_incomplete" {
t.Fatalf("warnings = %#v, want %#v", got, want) t.Fatalf("diagnostics = %#v, want validation-incomplete group", output.Diagnostics)
} }
if len(recorder.checkpoint.Outputs) != 0 || len(recorder.checkpoint.Rejected) != 0 { if len(recorder.checkpoint.Outputs) != 0 || len(recorder.checkpoint.Rejected) != 0 {
t.Fatalf("extract checkpoint = %#v, want no persisted incomplete output", recorder.checkpoint) t.Fatalf("extract checkpoint = %#v, want no persisted incomplete output", recorder.checkpoint)

View File

@@ -35,11 +35,10 @@ func (v *mutatingMetadataValidator) Validate(_ context.Context, request contract
return contracts.ValidationResult{Approved: true}, nil return contracts.ValidationResult{Approved: true}, nil
} }
func (r *extractCaptureRecorder) ExtractSucceeded(_ string, _ string, _ []CheckpointFingerprint, outputs []CheckpointArtifact, rejected []contracts.RejectedOutput, warnings []contracts.Warning) error { func (r *extractCaptureRecorder) ExtractSucceeded(_ string, _ string, _ []CheckpointFingerprint, outputs []CheckpointArtifact, rejected []contracts.RejectedOutput) error {
r.checkpoint = ExtractCheckpoint{ r.checkpoint = ExtractCheckpoint{
Outputs: cloneCheckpointArtifacts(outputs), Outputs: cloneCheckpointArtifacts(outputs),
Rejected: cloneRejectedOutputs(rejected), Rejected: cloneRejectedOutputs(rejected),
Warnings: cloneWarnings(warnings),
} }
return nil return nil
} }
@@ -56,7 +55,6 @@ func (l *extractResultLoader) Extract(string, string, []CheckpointFingerprint) (
return ExtractCheckpoint{ return ExtractCheckpoint{
Outputs: cloneCheckpointArtifacts(l.checkpoint.Outputs), Outputs: cloneCheckpointArtifacts(l.checkpoint.Outputs),
Rejected: cloneRejectedOutputs(l.checkpoint.Rejected), Rejected: cloneRejectedOutputs(l.checkpoint.Rejected),
Warnings: cloneWarnings(l.checkpoint.Warnings),
}, l.decision }, l.decision
} }
@@ -157,7 +155,13 @@ func TestRunnerContinuesFromFreshAndReusedExtractResults(t *testing.T) {
extractCalls++ extractCalls++
return erasedTypedResult{ return erasedTypedResult{
Value: typedValueForLane(0, request.Chunk.Index), Value: typedValueForLane(0, request.Chunk.Index),
Warnings: []contracts.Warning{{Scope: "extract", ReasonCode: "observed", Message: "accepted extract"}}, Diagnostics: []contracts.ProducerDiagnostic{{
Disposition: contracts.DiagnosticDispositionObservation,
Category: contracts.DiagnosticCategoryNormalization,
ReasonCode: "accepted_extract_normalized",
OccurrenceCount: 1,
Samples: []contracts.DiagnosticSample{{Scope: "extract", Message: "accepted extract"}},
}},
}, nil }, nil
}) })
@@ -206,12 +210,12 @@ func TestRunnerContinuesFromFreshAndReusedExtractResults(t *testing.T) {
if !reflect.DeepEqual(reused.NormalizeOutputs, fresh.NormalizeOutputs) { if !reflect.DeepEqual(reused.NormalizeOutputs, fresh.NormalizeOutputs) {
t.Fatalf("reused normalize outputs = %#v, want fresh outputs %#v", reused.NormalizeOutputs, fresh.NormalizeOutputs) t.Fatalf("reused normalize outputs = %#v, want fresh outputs %#v", reused.NormalizeOutputs, fresh.NormalizeOutputs)
} }
if !reflect.DeepEqual(reused.Warnings, fresh.Warnings) { if !reflect.DeepEqual(reused.Diagnostics, fresh.Diagnostics) {
t.Fatalf("reused warnings = %#v, want fresh warnings %#v", reused.Warnings, fresh.Warnings) t.Fatalf("reused diagnostics = %#v, want fresh diagnostics %#v", reused.Diagnostics, fresh.Diagnostics)
} }
} }
func TestRunnerPromotesOnlyAcceptedExtractRetryWarnings(t *testing.T) { func TestRunnerPromotesOnlyAcceptedExtractRetryDiagnostics(t *testing.T) {
prepared := preparedAttemptDebugPipeline(t) prepared := preparedAttemptDebugPipeline(t)
prepared.Steps[0].lanes[0].resolved.Extract.Retries = 1 prepared.Steps[0].lanes[0].resolved.Extract.Retries = 1
attempts := 0 attempts := 0
@@ -223,7 +227,7 @@ func TestRunnerPromotesOnlyAcceptedExtractRetryWarnings(t *testing.T) {
} }
return erasedTypedResult{ return erasedTypedResult{
Value: typedValueForLane(0, request.Chunk.Index), Value: typedValueForLane(0, request.Chunk.Index),
Warnings: []contracts.Warning{{Scope: scope, ReasonCode: "observed", Message: scope}}, Diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic(scope, scope)},
ModelCandidate: attemptCandidate(t, fmt.Sprintf(`{"items":["%s"]}`, scope)), ModelCandidate: attemptCandidate(t, fmt.Sprintf(`{"items":["%s"]}`, scope)),
}, nil }, nil
}) })
@@ -240,8 +244,8 @@ func TestRunnerPromotesOnlyAcceptedExtractRetryWarnings(t *testing.T) {
if attempts != 2 { if attempts != 2 {
t.Fatalf("extract attempts = %d, want 2", attempts) t.Fatalf("extract attempts = %d, want 2", attempts)
} }
if len(output.Warnings) != 1 || output.Warnings[0].Scope != "accepted" { if len(output.Diagnostics.Groups) != 1 || output.Diagnostics.Groups[0].Samples[0].Scope != "accepted" {
t.Fatalf("promoted warnings = %#v, want accepted attempt only", output.Warnings) t.Fatalf("promoted diagnostics = %#v, want accepted attempt only", output.Diagnostics)
} }
assertAttemptEnvelopeSequence(t, debug, "extract/notes/chunk-000001", 1, 2) assertAttemptEnvelopeSequence(t, debug, "extract/notes/chunk-000001", 1, 2)
first := debug.envelope(t, "extract/notes/chunk-000001/attempt-01.json") first := debug.envelope(t, "extract/notes/chunk-000001/attempt-01.json")

View File

@@ -17,7 +17,7 @@ func TestRunnerHandlesRetryableNormalizeFallbacks(t *testing.T) {
validator *preparedValidator validator *preparedValidator
wantCalls int wantCalls int
wantItem string wantItem string
wantWarnings []string wantMessages []string
wantRejected int wantRejected int
wantDebug []string wantDebug []string
wantCheckpoint int wantCheckpoint int
@@ -30,7 +30,7 @@ func TestRunnerHandlesRetryableNormalizeFallbacks(t *testing.T) {
}, },
wantCalls: 1, wantCalls: 1,
wantItem: "fallback", wantItem: "fallback",
wantWarnings: []string{"ordinary", "fallback-warning"}, wantMessages: []string{"ordinary", "fallback-warning"},
wantDebug: []string{`"another_attempt":false`, `"fallback_accepted":true`}, wantDebug: []string{`"another_attempt":false`, `"fallback_accepted":true`},
wantCheckpoint: 1, wantCheckpoint: 1,
}, },
@@ -41,11 +41,11 @@ func TestRunnerHandlesRetryableNormalizeFallbacks(t *testing.T) {
if attempt == 1 { if attempt == 1 {
return retryableNormalizeResult("discarded", "discarded-ordinary", "discarded-fallback") return retryableNormalizeResult("discarded", "discarded-ordinary", "discarded-fallback")
} }
return erasedTypedResult{Value: codecNotes{Items: []string{"accepted"}}, Warnings: []contracts.Warning{{Scope: "accepted", ReasonCode: "ordinary", Message: "accepted-warning"}}} return erasedTypedResult{Value: codecNotes{Items: []string{"accepted"}}, Diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic("accepted", "accepted-warning")}}
}, },
wantCalls: 2, wantCalls: 2,
wantItem: "accepted", wantItem: "accepted",
wantWarnings: []string{"accepted-warning"}, wantMessages: []string{"accepted-warning"},
wantDebug: []string{`"another_attempt":true`, `"fallback_accepted":false`}, wantDebug: []string{`"another_attempt":true`, `"fallback_accepted":false`},
wantCheckpoint: 1, wantCheckpoint: 1,
}, },
@@ -57,7 +57,7 @@ func TestRunnerHandlesRetryableNormalizeFallbacks(t *testing.T) {
}, },
wantCalls: 2, wantCalls: 2,
wantItem: "fallback-2", wantItem: "fallback-2",
wantWarnings: []string{"ordinary-2", "fallback-warning-2"}, wantMessages: []string{"ordinary-2", "fallback-warning-2"},
wantDebug: []string{`"another_attempt":false`, `"fallback_accepted":true`}, wantDebug: []string{`"another_attempt":false`, `"fallback_accepted":true`},
wantCheckpoint: 1, wantCheckpoint: 1,
}, },
@@ -140,17 +140,73 @@ func TestRunnerHandlesRetryableNormalizeFallbacks(t *testing.T) {
if got := firstNote(normalized); got != tc.wantItem { if got := firstNote(normalized); got != tc.wantItem {
t.Fatalf("normalized item = %q, want %q", got, tc.wantItem) t.Fatalf("normalized item = %q, want %q", got, tc.wantItem)
} }
gotWarnings := make([]string, len(output.Warnings)) var gotMessages []string
for index, warning := range output.Warnings { for _, group := range output.Diagnostics.Groups {
gotWarnings[index] = warning.Message for _, sample := range group.Samples {
gotMessages = append(gotMessages, sample.Message)
} }
if strings.Join(gotWarnings, "|") != strings.Join(tc.wantWarnings, "|") { }
t.Fatalf("durable warnings = %#v, want %#v", gotWarnings, tc.wantWarnings) if strings.Join(gotMessages, "|") != strings.Join(tc.wantMessages, "|") {
t.Fatalf("durable diagnostics = %#v, want %#v", gotMessages, tc.wantMessages)
} }
}) })
} }
} }
func TestRunnerForwardsModuleRequestedNormalizeCorrection(t *testing.T) {
const (
defective = `{"duplicate_groups":[{"candidate_numbers":[1,99],"canonical_candidate_number":1}]}`
guidance = "Duplicate group 1 must use only supplied candidate numbers. Return one complete corrected response."
)
prepared := preparedAttemptDebugPipeline(t)
lane := &prepared.Steps[0].lanes[0]
lane.resolved.Normalize.Retries = 1
var observed *contracts.SemanticCorrection
calls := 0
lane.typed.normalize = func(_ context.Context, _ any, request contracts.TypedNormalizeRequest[any]) (erasedTypedResult, error) {
calls++
if request.Correction != nil {
clone, err := contracts.CloneSemanticCorrection(request.Correction)
if err != nil {
return erasedTypedResult{}, err
}
observed = clone
}
if calls == 1 {
return erasedTypedResult{
Value: codecNotes{Items: []string{"safe fallback"}},
ModelCandidate: attemptCandidate(t, defective),
Retry: &contracts.NormalizeRetry{
ReasonCode: "semantic_proposal_invalid",
Message: "operator-facing proposal diagnostic",
CorrectionGuidance: guidance,
},
}, nil
}
return erasedTypedResult{Value: codecNotes{Items: []string{"corrected"}}}, nil
}
debug := newCapturedDebugRecorder()
output, err := New().Run(context.Background(), RunInput{Prepared: prepared, RawInput: []byte("input"), Debug: debug})
if err != nil {
t.Fatalf("Run() error = %v", err)
}
if calls != 2 || observed == nil || string(observed.AssistantResponse) != defective || observed.UserGuidance != guidance {
t.Fatalf("normalize calls = %d correction = %#v", calls, observed)
}
if len(output.NormalizeOutputs) != 1 || len(output.Rejected) != 0 {
t.Fatalf("run output = %#v, want corrected accepted output", output)
}
var retryDebug strings.Builder
for _, name := range debug.names() {
if strings.HasPrefix(name, "normalize/notes/attempt-") && strings.HasSuffix(name, ".json") {
retryDebug.Write(debug.json[name])
}
}
if !strings.Contains(retryDebug.String(), `"correction_available":true`) || strings.Contains(retryDebug.String(), guidance) || strings.Contains(retryDebug.String(), defective) {
t.Fatalf("retry debug = %s, want safe correction metadata without content", retryDebug.String())
}
}
func TestRunnerValidatesNormalizeRetryDiagnostics(t *testing.T) { func TestRunnerValidatesNormalizeRetryDiagnostics(t *testing.T) {
const ( const (
reasonSentinel = "reason-diagnostic-sentinel" reasonSentinel = "reason-diagnostic-sentinel"
@@ -158,9 +214,11 @@ func TestRunnerValidatesNormalizeRetryDiagnostics(t *testing.T) {
) )
reasonOverLimit := strings.Repeat("r", contracts.MaxNormalizeRetryReasonCodeBytes-len(reasonSentinel)) + reasonSentinel + "x" reasonOverLimit := strings.Repeat("r", contracts.MaxNormalizeRetryReasonCodeBytes-len(reasonSentinel)) + reasonSentinel + "x"
messageOverLimit := strings.Repeat("m", contracts.MaxNormalizeRetryMessageBytes-len(messageSentinel)) + messageSentinel + "x" messageOverLimit := strings.Repeat("m", contracts.MaxNormalizeRetryMessageBytes-len(messageSentinel)) + messageSentinel + "x"
guidanceOverLimit := strings.Repeat("g", contracts.MaxNormalizeRetryCorrectionGuidanceBytes-len(messageSentinel)) + messageSentinel + "x"
tests := []struct { tests := []struct {
name string name string
retry contracts.NormalizeRetry retry contracts.NormalizeRetry
candidate *contracts.ModelCandidate
wantError string wantError string
hiddenValues []string hiddenValues []string
}{ }{
@@ -171,6 +229,15 @@ func TestRunnerValidatesNormalizeRetryDiagnostics(t *testing.T) {
Message: strings.Repeat("m", contracts.MaxNormalizeRetryMessageBytes), Message: strings.Repeat("m", contracts.MaxNormalizeRetryMessageBytes),
}, },
}, },
{
name: "accepts correction guidance byte limit with candidate",
retry: contracts.NormalizeRetry{
ReasonCode: reasonSentinel,
Message: messageSentinel,
CorrectionGuidance: strings.Repeat("g", contracts.MaxNormalizeRetryCorrectionGuidanceBytes),
},
candidate: attemptCandidate(t, `{"duplicate_groups":[]}`),
},
{ {
name: "rejects oversized reason code", name: "rejects oversized reason code",
retry: contracts.NormalizeRetry{ retry: contracts.NormalizeRetry{
@@ -207,6 +274,46 @@ func TestRunnerValidatesNormalizeRetryDiagnostics(t *testing.T) {
wantError: "message has invalid UTF-8", wantError: "message has invalid UTF-8",
hiddenValues: []string{reasonSentinel, messageSentinel}, hiddenValues: []string{reasonSentinel, messageSentinel},
}, },
{
name: "rejects oversized correction guidance",
retry: contracts.NormalizeRetry{
ReasonCode: reasonSentinel,
Message: messageSentinel,
CorrectionGuidance: guidanceOverLimit,
},
wantError: "correction guidance exceeds maximum length",
hiddenValues: []string{reasonSentinel, messageSentinel},
},
{
name: "rejects invalid correction guidance UTF-8",
retry: contracts.NormalizeRetry{
ReasonCode: reasonSentinel,
Message: messageSentinel,
CorrectionGuidance: messageSentinel + string([]byte{0xff}),
},
wantError: "correction guidance has invalid UTF-8",
hiddenValues: []string{reasonSentinel, messageSentinel},
},
{
name: "rejects blank correction guidance",
retry: contracts.NormalizeRetry{
ReasonCode: reasonSentinel,
Message: messageSentinel,
CorrectionGuidance: " \t\n ",
},
wantError: "correction guidance is blank",
hiddenValues: []string{reasonSentinel, messageSentinel},
},
{
name: "rejects correction guidance without candidate",
retry: contracts.NormalizeRetry{
ReasonCode: reasonSentinel,
Message: messageSentinel,
CorrectionGuidance: "Return one complete corrected response.",
},
wantError: "requires a model candidate",
hiddenValues: []string{reasonSentinel, messageSentinel},
},
{ {
name: "rejects blank reason code", name: "rejects blank reason code",
retry: contracts.NormalizeRetry{ retry: contracts.NormalizeRetry{
@@ -231,7 +338,7 @@ func TestRunnerValidatesNormalizeRetryDiagnostics(t *testing.T) {
t.Run(tc.name, func(t *testing.T) { t.Run(tc.name, func(t *testing.T) {
prepared := preparedAttemptDebugPipeline(t) prepared := preparedAttemptDebugPipeline(t)
prepared.Steps[0].lanes[0].typed.normalize = func(context.Context, any, contracts.TypedNormalizeRequest[any]) (erasedTypedResult, error) { prepared.Steps[0].lanes[0].typed.normalize = func(context.Context, any, contracts.TypedNormalizeRequest[any]) (erasedTypedResult, error) {
return erasedTypedResult{Value: codecNotes{Items: []string{"safe"}}, Retry: &tc.retry}, nil return erasedTypedResult{Value: codecNotes{Items: []string{"safe"}}, Retry: &tc.retry, ModelCandidate: tc.candidate}, nil
} }
debug := newCapturedDebugRecorder() debug := newCapturedDebugRecorder()
output, err := New().Run(context.Background(), RunInput{Prepared: prepared, RawInput: []byte("input"), Debug: debug}) output, err := New().Run(context.Background(), RunInput{Prepared: prepared, RawInput: []byte("input"), Debug: debug})
@@ -263,11 +370,11 @@ func TestRunnerValidatesNormalizeRetryDiagnostics(t *testing.T) {
func retryableNormalizeResult(item, ordinary, fallback string) erasedTypedResult { func retryableNormalizeResult(item, ordinary, fallback string) erasedTypedResult {
return erasedTypedResult{ return erasedTypedResult{
Value: codecNotes{Items: []string{item}}, Value: codecNotes{Items: []string{item}},
Warnings: []contracts.Warning{{Scope: "attempt", ReasonCode: "ordinary", Message: ordinary}}, Diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic("attempt", ordinary)},
Retry: &contracts.NormalizeRetry{ Retry: &contracts.NormalizeRetry{
ReasonCode: "retryable_normalization", ReasonCode: "retryable_normalization",
Message: "safe fallback is available", Message: "safe fallback is available",
FallbackWarnings: []contracts.Warning{{Scope: "fallback", ReasonCode: "fallback", Message: fallback}}, FallbackDiagnostics: []contracts.ProducerDiagnostic{producerDiagnostic("fallback", fallback)},
}, },
} }
} }

View File

@@ -11,19 +11,19 @@ import (
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts" "gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
) )
type warningChunker struct { type diagnosticChunker struct {
key string key string
plan source.ChunkPlan plan source.ChunkPlan
calls int calls int
} }
func (c *warningChunker) Key() string { return c.key } func (c *diagnosticChunker) Key() string { return c.key }
func (*warningChunker) ReferenceSlots() []contracts.ReferenceSlot { return nil } func (*diagnosticChunker) ReferenceSlots() []contracts.ReferenceSlot { return nil }
func (c *warningChunker) Plan(context.Context, contracts.ChunkRequest) (contracts.ChunkPlanResult, error) { func (c *diagnosticChunker) Plan(context.Context, contracts.ChunkRequest) (contracts.ChunkPlanResult, error) {
c.calls++ c.calls++
return contracts.ChunkPlanResult{ return contracts.ChunkPlanResult{
Plan: source.CloneChunkPlan(c.plan), Plan: source.CloneChunkPlan(c.plan),
Warnings: []contracts.Warning{{Scope: fmt.Sprintf("operation-%d", c.calls), ReasonCode: "operation", Message: "operation warning"}}, Diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic(fmt.Sprintf("operation-%d", c.calls), "operation diagnostic")},
}, nil }, nil
} }
@@ -40,14 +40,14 @@ func (v chunkValidationFunc) Validate(ctx context.Context, request contracts.Chu
return v.validate(ctx, request) return v.validate(ctx, request)
} }
func TestRunnerPromotesOnlyTerminalRejectionWarnings(t *testing.T) { func TestRunnerPromotesOnlyTerminalRejectionDiagnostics(t *testing.T) {
for _, target := range []ModuleStage{StageChunk, StageExtract, StageMerge, StageNormalize} { for _, target := range []ModuleStage{StageChunk, StageExtract, StageMerge, StageNormalize} {
t.Run(string(target), func(t *testing.T) { t.Run(string(target), func(t *testing.T) {
prepared := preparedAttemptDebugPipeline(t) prepared := preparedAttemptDebugPipeline(t)
lane := &prepared.Steps[0].lanes[0] lane := &prepared.Steps[0].lanes[0]
attempts := 0 attempts := 0
first := func() contracts.ValidationResult { first := func() contracts.ValidationResult {
return contracts.ValidationResult{Approved: true, Warnings: []contracts.Warning{{Scope: fmt.Sprintf("validator-%d", attempts), ReasonCode: "validator", Message: "validator warning"}}} return contracts.ValidationResult{Approved: true, Diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic(fmt.Sprintf("validator-%d", attempts), "validator diagnostic")}}
} }
reject := func() contracts.ValidationResult { reject := func() contracts.ValidationResult {
return contracts.ValidationResult{Approved: false, ReasonCode: "rejected", Message: "rejected", CorrectionGuidance: "return an acceptable candidate"} return contracts.ValidationResult{Approved: false, ReasonCode: "rejected", Message: "rejected", CorrectionGuidance: "return an acceptable candidate"}
@@ -59,7 +59,7 @@ func TestRunnerPromotesOnlyTerminalRejectionWarnings(t *testing.T) {
case StageChunk: case StageChunk:
prepared.resolved.ChunkValidationPolicy.SemanticRejection = SemanticRejectionRejectOutput prepared.resolved.ChunkValidationPolicy.SemanticRejection = SemanticRejectionRejectOutput
chunker := prepared.chunker.(*typedTestChunker) chunker := prepared.chunker.(*typedTestChunker)
prepared.chunker = &warningChunker{key: prepared.resolved.Chunk.Module, plan: source.CloneChunkPlan(chunker.plan)} prepared.chunker = &diagnosticChunker{key: prepared.resolved.Chunk.Module, plan: source.CloneChunkPlan(chunker.plan)}
prepared.resolved.Chunk.Retries = 1 prepared.resolved.Chunk.Retries = 1
prepared.chunkValidators.validators = []preparedValidator{ prepared.chunkValidators.validators = []preparedValidator{
{resolved: ResolvedValidator{Binding: Binding("warning-approval"), Target: ValidatorTargetChunk}, chunk: chunkValidationFunc{name: "warning-approval", validate: func(context.Context, contracts.ChunkValidationRequest) (contracts.ValidationResult, error) { {resolved: ResolvedValidator{Binding: Binding("warning-approval"), Target: ValidatorTargetChunk}, chunk: chunkValidationFunc{name: "warning-approval", validate: func(context.Context, contracts.ChunkValidationRequest) (contracts.ValidationResult, error) {
@@ -69,34 +69,34 @@ func TestRunnerPromotesOnlyTerminalRejectionWarnings(t *testing.T) {
return reject(), nil return reject(), nil
}}}, }}},
} }
chunkerWithWarnings := prepared.chunker.(*warningChunker) chunkerWithDiagnostics := prepared.chunker.(*diagnosticChunker)
first = func() contracts.ValidationResult { first = func() contracts.ValidationResult {
return contracts.ValidationResult{Approved: true, Warnings: []contracts.Warning{{Scope: fmt.Sprintf("validator-%d", chunkerWithWarnings.calls), ReasonCode: "validator", Message: "validator warning"}}} return contracts.ValidationResult{Approved: true, Diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic(fmt.Sprintf("validator-%d", chunkerWithDiagnostics.calls), "validator diagnostic")}}
} }
case StageExtract: case StageExtract:
lane.resolved.Extract.Retries = 1 lane.resolved.Extract.Retries = 1
lane.resolved.ExtractValidationPolicy.SemanticRejection = SemanticRejectionRejectOutput lane.resolved.ExtractValidationPolicy.SemanticRejection = SemanticRejectionRejectOutput
installExtractOperation(prepared, 0, func(context.Context, contracts.TypedExtractionRequest) (erasedTypedResult, error) { installExtractOperation(prepared, 0, func(context.Context, contracts.TypedExtractionRequest) (erasedTypedResult, error) {
attempts++ attempts++
return erasedTypedResult{Value: codecNotes{Items: []string{"extract"}}, Warnings: []contracts.Warning{{Scope: fmt.Sprintf("operation-%d", attempts), ReasonCode: "operation", Message: "operation warning"}}, ModelCandidate: attemptCandidate(t, fmt.Sprintf(`{"items":["%d"]}`, attempts))}, nil return erasedTypedResult{Value: codecNotes{Items: []string{"extract"}}, Diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic(fmt.Sprintf("operation-%d", attempts), "operation diagnostic")}, ModelCandidate: attemptCandidate(t, fmt.Sprintf(`{"items":["%d"]}`, attempts))}, nil
}) })
lane.extractValidators.validators = rejectionWarningTypedValidators(first, reject) lane.extractValidators.validators = rejectionDiagnosticTypedValidators(first, reject)
case StageMerge: case StageMerge:
lane.resolved.Merge.Retries = 1 lane.resolved.Merge.Retries = 1
lane.resolved.MergeValidationPolicy.SemanticRejection = SemanticRejectionRejectOutput lane.resolved.MergeValidationPolicy.SemanticRejection = SemanticRejectionRejectOutput
lane.typed.merge = func(context.Context, any, contracts.TypedMergeRequest[any]) (erasedTypedResult, error) { lane.typed.merge = func(context.Context, any, contracts.TypedMergeRequest[any]) (erasedTypedResult, error) {
attempts++ attempts++
return erasedTypedResult{Value: codecNotes{Items: []string{"merge"}}, Warnings: []contracts.Warning{{Scope: fmt.Sprintf("operation-%d", attempts), ReasonCode: "operation", Message: "operation warning"}}, ModelCandidate: attemptCandidate(t, fmt.Sprintf(`{"items":["%d"]}`, attempts))}, nil return erasedTypedResult{Value: codecNotes{Items: []string{"merge"}}, Diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic(fmt.Sprintf("operation-%d", attempts), "operation diagnostic")}, ModelCandidate: attemptCandidate(t, fmt.Sprintf(`{"items":["%d"]}`, attempts))}, nil
} }
lane.mergeValidators.validators = rejectionWarningTypedValidators(first, reject) lane.mergeValidators.validators = rejectionDiagnosticTypedValidators(first, reject)
case StageNormalize: case StageNormalize:
lane.resolved.Normalize.Retries = 1 lane.resolved.Normalize.Retries = 1
lane.resolved.NormalizeValidationPolicy.SemanticRejection = SemanticRejectionRejectOutput lane.resolved.NormalizeValidationPolicy.SemanticRejection = SemanticRejectionRejectOutput
lane.typed.normalize = func(context.Context, any, contracts.TypedNormalizeRequest[any]) (erasedTypedResult, error) { lane.typed.normalize = func(context.Context, any, contracts.TypedNormalizeRequest[any]) (erasedTypedResult, error) {
attempts++ attempts++
return erasedTypedResult{Value: codecNotes{Items: []string{"normalize"}}, Warnings: []contracts.Warning{{Scope: fmt.Sprintf("operation-%d", attempts), ReasonCode: "operation", Message: "operation warning"}}, ModelCandidate: attemptCandidate(t, fmt.Sprintf(`{"items":["%d"]}`, attempts))}, nil return erasedTypedResult{Value: codecNotes{Items: []string{"normalize"}}, Diagnostics: []contracts.ProducerDiagnostic{producerDiagnostic(fmt.Sprintf("operation-%d", attempts), "operation diagnostic")}, ModelCandidate: attemptCandidate(t, fmt.Sprintf(`{"items":["%d"]}`, attempts))}, nil
} }
lane.normalizeValidators.validators = rejectionWarningTypedValidators(first, reject) lane.normalizeValidators.validators = rejectionDiagnosticTypedValidators(first, reject)
} }
output, err := New().Run(context.Background(), RunInput{Prepared: prepared, RawInput: []byte("input"), Checkpoints: recorder, Debug: debug}) output, err := New().Run(context.Background(), RunInput{Prepared: prepared, RawInput: []byte("input"), Checkpoints: recorder, Debug: debug})
@@ -109,29 +109,26 @@ func TestRunnerPromotesOnlyTerminalRejectionWarnings(t *testing.T) {
wantScopes = []string{"operation-1", "validator-1"} wantScopes = []string{"operation-1", "validator-1"}
wantAttempts = 1 wantAttempts = 1
} }
if got := rejectionWarningScopes(output.Warnings); !reflect.DeepEqual(got, wantScopes) { if got := rejectionDiagnosticScopes(output.Diagnostics.Groups); !reflect.DeepEqual(got, wantScopes) {
t.Fatalf("published warning scopes = %#v, want %#v", got, wantScopes) t.Fatalf("published diagnostic scopes = %#v, want %#v", got, wantScopes)
} }
if len(output.Rejected) != 1 || output.Rejected[0].AttemptCount != wantAttempts { if len(output.Rejected) != 1 || output.Rejected[0].AttemptCount != wantAttempts {
t.Fatalf("rejections = %#v, want terminal rejection after %d attempt(s)", output.Rejected, wantAttempts) t.Fatalf("rejections = %#v, want terminal rejection after %d attempt(s)", output.Rejected, wantAttempts)
} }
if target == StageExtract && !reflect.DeepEqual(rejectionWarningScopes(recorder.checkpoint.Warnings), wantScopes) {
t.Fatalf("extract checkpoint warnings = %#v, want %#v", recorder.checkpoint.Warnings, wantScopes)
}
attemptPath := fmt.Sprintf("%s/notes/attempt-01.json", target) attemptPath := fmt.Sprintf("%s/notes/attempt-01.json", target)
if target == StageChunk { if target == StageChunk {
attemptPath = "chunk/attempt-01.json" attemptPath = "chunk/attempt-01.json"
} else if target == StageExtract { } else if target == StageExtract {
attemptPath = "extract/notes/chunk-000001/attempt-01.json" attemptPath = "extract/notes/chunk-000001/attempt-01.json"
} }
if !strings.Contains(string(debug.json[attemptPath]), "operation-1") || !strings.Contains(string(debug.json[attemptPath]), "validator-1") { if !strings.Contains(string(debug.json[attemptPath]), "rejection") {
t.Fatalf("first attempt debug = %s, want discarded warnings", debug.json[attemptPath]) t.Fatalf("first attempt debug = %s, want rejection", debug.json[attemptPath])
} }
}) })
} }
} }
func rejectionWarningTypedValidators(first func() contracts.ValidationResult, reject func() contracts.ValidationResult) []preparedValidator { func rejectionDiagnosticTypedValidators(first func() contracts.ValidationResult, reject func() contracts.ValidationResult) []preparedValidator {
return []preparedValidator{ return []preparedValidator{
{resolved: ResolvedValidator{Binding: Binding("warning-approval"), Target: ValidatorTargetTyped, ArtifactKind: "test/notes"}, typedValidate: func(context.Context, any, typedValidationTarget) (contracts.ValidationResult, error) { {resolved: ResolvedValidator{Binding: Binding("warning-approval"), Target: ValidatorTargetTyped, ArtifactKind: "test/notes"}, typedValidate: func(context.Context, any, typedValidationTarget) (contracts.ValidationResult, error) {
return first(), nil return first(), nil
@@ -142,10 +139,10 @@ func rejectionWarningTypedValidators(first func() contracts.ValidationResult, re
} }
} }
func rejectionWarningScopes(warnings []contracts.Warning) []string { func rejectionDiagnosticScopes(groups []contracts.DiagnosticGroup) []string {
scopes := make([]string, len(warnings)) scopes := make([]string, len(groups))
for index := range warnings { for index := range groups {
scopes[index] = warnings[index].Scope scopes[index] = groups[index].Samples[0].Scope
} }
return scopes return scopes
} }

View File

@@ -61,7 +61,7 @@ func (s *reuseLineageCheckpointSpy) ExtractRunning(laneID, _ string, _ []Checkpo
return nil return nil
} }
func (s *reuseLineageCheckpointSpy) ExtractSucceeded(laneID, _ string, _ []CheckpointFingerprint, _ []CheckpointArtifact, _ []contracts.RejectedOutput, _ []contracts.Warning) error { func (s *reuseLineageCheckpointSpy) ExtractSucceeded(laneID, _ string, _ []CheckpointFingerprint, _ []CheckpointArtifact, _ []contracts.RejectedOutput) error {
s.write("extract", "succeeded", laneID) s.write("extract", "succeeded", laneID)
return nil return nil
} }
@@ -76,7 +76,7 @@ func (s *reuseLineageCheckpointSpy) MergeRunning(laneID, _ string, _ []Checkpoin
return nil return nil
} }
func (s *reuseLineageCheckpointSpy) MergeSucceeded(laneID, _ string, _ []CheckpointFingerprint, _ CheckpointArtifact, _ []contracts.Warning) error { func (s *reuseLineageCheckpointSpy) MergeSucceeded(laneID, _ string, _ []CheckpointFingerprint, _ CheckpointArtifact) error {
s.write("merge", "succeeded", laneID) s.write("merge", "succeeded", laneID)
return nil return nil
} }
@@ -96,7 +96,7 @@ func (s *reuseLineageCheckpointSpy) NormalizeRunning(laneID, _ string, _ []Check
return nil return nil
} }
func (s *reuseLineageCheckpointSpy) NormalizeSucceeded(laneID, _ string, _ []CheckpointFingerprint, _ CheckpointArtifact, _ []contracts.Warning) error { func (s *reuseLineageCheckpointSpy) NormalizeSucceeded(laneID, _ string, _ []CheckpointFingerprint, _ CheckpointArtifact) error {
s.write("normalize", "succeeded", laneID) s.write("normalize", "succeeded", laneID)
return nil return nil
} }

View File

@@ -120,7 +120,7 @@ func TestRunnerContinuesNormalizeAfterValidatorFailure(t *testing.T) {
if output.Manifest.ValidationStatus != "incomplete" || !containsValidationSummary(output.Manifest.ValidationSummaries, "incomplete") { if output.Manifest.ValidationStatus != "incomplete" || !containsValidationSummary(output.Manifest.ValidationSummaries, "incomplete") {
t.Fatalf("manifest validation = %#v, want incomplete normalization provenance", output.Manifest) t.Fatalf("manifest validation = %#v, want incomplete normalization provenance", output.Manifest)
} }
if got, want := output.Warnings, []contracts.Warning{{Scope: "unavailable", ReasonCode: "validator_execution_incomplete", Message: "Validator execution did not complete within its configured budget."}}; !containsWarnings(got, want) { if len(output.Diagnostics.Groups) != 1 || output.Diagnostics.Groups[0].ReasonCode != "validator_execution_incomplete" {
t.Fatalf("warnings = %#v, want %#v", got, want) t.Fatalf("diagnostics = %#v, want validation-incomplete group", output.Diagnostics)
} }
} }

Some files were not shown because too many files have changed in this diff Show More