17 Commits

Author SHA1 Message Date
77ae132e56 Prepare the v0.6.0 release 2026-08-29 12:46:18 +00:00
deebc89255 Make command line references pipeline scoped 2026-08-29 12:34:31 +00:00
f208dbe954 Make combat scene validation more reliable 2026-08-29 11:39:02 +00:00
7e626753bf Clarify semantic reconciliation candidate numbers 2026-08-29 03:16:24 +00:00
917d150279 Improve D&D validation reliability 2026-08-29 01:24:45 +00:00
4da9360d74 Improve semantic reconciliation retries 2026-08-28 18:25:00 +00:00
5cab4e512e Improve item occurrence holder corrections 2026-08-28 14:37:22 +00:00
b178f1c684 Repair reversed D&D evidence ranges 2026-08-28 13:45:21 +00:00
a2610757cd Polish combat scene validation 2026-08-28 02:26:14 +00:00
3ad34956c5 Complete combat semantics validator verification 2026-08-28 00:24:35 +00:00
22e6caa2a0 Document combat semantics validator evaluation 2026-08-28 00:23:50 +00:00
9cb7462800 Test combat semantics correction flow 2026-08-28 00:22:56 +00:00
87385b7e14 Register optional combat semantics validator 2026-08-28 00:20:59 +00:00
0ae5ea7637 Implement combat semantics validator 2026-08-28 00:19:49 +00:00
6bc883dfb6 Add combat semantics validator prompt assets 2026-08-28 00:16:23 +00:00
fcfff3ad15 Extract shared D&D combat policy 2026-08-28 00:13:44 +00:00
0f2b23dce1 Plan combat scene validation 2026-08-27 23:26:59 +00:00
157 changed files with 4148 additions and 1144 deletions

View File

@@ -16,7 +16,8 @@
"type": "string" "type": "string"
}, },
"turn_kind": { "turn_kind": {
"type": "string" "type": "string",
"enum": ["turn", "reaction", "legendary_action", "lair_action", "other"]
}, },
"source_refs": { "source_refs": {
"type": "array", "type": "array",

View File

@@ -13,7 +13,10 @@
"required": ["name", "kind", "source_refs"], "required": ["name", "kind", "source_refs"],
"properties": { "properties": {
"name": {"type": "string"}, "name": {"type": "string"},
"kind": {"type": "string"}, "kind": {
"type": "string",
"enum": ["engaged", "killed", "fled", "captured", "incapacitated"]
},
"source_refs": { "source_refs": {
"type": "array", "type": "array",
"items": { "items": {

View File

@@ -18,12 +18,20 @@ when the transcript explicitly describes it being physically destroyed or
expended as a non-payment component. Use `transferred` only when possession expended as a non-payment component. Use `transferred` only when possession
moves between two distinct named party members. moves between two distinct named party members.
Return both `from` and `to` for every occurrence, using `null` when a holder does not Return both `from` and `to` for every occurrence. Use JSON `null`, not an empty
apply. For `discovered`, set both holders to `null`. For `acquired`, set `from` string, whenever a holder does not apply. Follow this holder matrix exactly:
to `null` and provide `to`; for `lost` and `consumed`, provide `from` and set
`to` to `null`; and for `transferred`, provide both holders. Use `party` only | `kind` | required `from` | required `to` |
for collective or unresolved party possession, never for either side of a | --- | --- | --- |
transfer. Do not emit a transfer for a gift, sale, or payment outside the party. | `discovered` | `null` | `null` |
| `acquired` | `null` | `party` or the named party member gaining possession |
| `lost` | `party` or the named party member losing possession | `null` |
| `consumed` | `party` or the named party member consuming the item | `null` |
| `transferred` | one named party member | a different named party member |
Use `party` only for collective or unresolved party possession, never for
either side of a transfer. Do not emit a transfer for a gift, sale, or payment
outside the party.
Ordinary non-depleting use is not an occurrence. Do not infer acquisition from a Ordinary non-depleting use is not an occurrence. Do not infer acquisition from a
discovery, or discovery from an acquisition: emit both only when each is discovery, or discovery from an acquisition: emit both only when each is

View File

@@ -13,7 +13,10 @@
"required": ["name", "kind", "quantity", "from", "to", "source_refs"], "required": ["name", "kind", "quantity", "from", "to", "source_refs"],
"properties": { "properties": {
"name": {"type": "string"}, "name": {"type": "string"},
"kind": {"type": "string"}, "kind": {
"type": "string",
"enum": ["discovered", "acquired", "lost", "consumed", "transferred"]
},
"quantity": {"type": ["integer", "null"]}, "quantity": {"type": ["integer", "null"]},
"from": {"type": ["string", "null"]}, "from": {"type": ["string", "null"]},
"to": {"type": ["string", "null"]}, "to": {"type": ["string", "null"]},

View File

@@ -5,5 +5,5 @@ evidence, similar objects, or a shared owner as sufficient.
Keep currency denominations and materially different item types separate. Keep Keep currency denominations and materially different item types separate. Keep
uncertain aliases separate. Do not infer an item property or uniqueness. uncertain aliases separate. Do not infer an item property or uniqueness.
When selecting a canonical display name, choose one supplied candidate name Set `canonical_candidate_number` to the supplied candidate number whose label
that is the clearest established designation. is the clearest established designation.

View File

@@ -5,4 +5,5 @@ nearby evidence, nested places, or generic labels as sufficient.
Keep parent and child places separate, as well as similarly named places and Keep parent and child places separate, as well as similarly named places and
uncertain aliases. uncertain aliases.
When selecting a canonical display name, prefer the clearest established name. Set `canonical_candidate_number` to the supplied candidate number whose label
is the clearest established name.

View File

@@ -16,7 +16,8 @@
"type": "string" "type": "string"
}, },
"kind": { "kind": {
"type": "string" "type": "string",
"enum": ["mentioned", "noncombat_presence", "dialogue", "combat_ally", "combat_opponent", "other"]
}, },
"source_refs": { "source_refs": {
"type": "array", "type": "array",

View File

@@ -3,7 +3,8 @@ contextual labels and cited transcript windows. Preserve distinct individuals
even when their names are similar or their contextual descriptions are even when their names are similar or their contextual descriptions are
identical. identical.
When selecting a canonical display name, prefer a complete, stable proper name Set `canonical_candidate_number` to the supplied candidate number whose label
is the preferred canonical display name. Prefer a complete, stable proper name
over an abbreviation. Prefer an unadorned proper name over that name plus a over an abbreviation. Prefer an unadorned proper name over that name plus a
contextual class, role, title, or relationship descriptor unless the transcript contextual class, role, title, or relationship descriptor unless the transcript
establishes the descriptor as part of the person's name. A longer display name establishes the descriptor as part of the person's name. A longer display name

View File

@@ -5,9 +5,7 @@ into multiple scenes or use facts that are not supported by it.
Return one kind, one concise title, and one concise summary. Choose exactly one Return one kind, one concise title, and one concise summary. Choose exactly one
kind: kind:
- combat: active combat materially organizes the scene, including - combat: a scene classified as combat under the shared combat policy.
initiative-like exchanges or sustained hostile action. Planning a fight or
discussing a completed fight is not combat by itself.
- narrative: current-session in-world play that is not principally active - narrative: current-session in-world play that is not principally active
combat, a prior-session recap, or sustained out-of-character session combat, a prior-session recap, or sustained out-of-character session
discussion. This includes exploration, travel, dialogue, investigation, discussion. This includes exploration, travel, dialogue, investigation,
@@ -20,15 +18,13 @@ kind:
play. play.
Narrative is the default for actual current-session gameplay that does not meet Narrative is the default for actual current-session gameplay that does not meet
another definition. When the accepted chunk is mixed: another definition. When the accepted chunk has no substantive active combat:
1. use combat when active combat is a substantive central activity, even with 1. use recap when recounting a previous session is the chunk's
brief setup, rules clarification, or immediate aftermath;
2. otherwise use recap when recounting a previous session is the chunk's
primary table purpose; primary table purpose;
3. otherwise use meta when sustained out-of-character session discussion is 2. otherwise use meta when sustained out-of-character session discussion is
primary and in-world progression is no more than incidental; and primary and in-world progression is no more than incidental; and
4. use narrative for all remaining current-session in-world play. 3. use narrative for all remaining current-session in-world play.
Brief table talk, dice resolution, rules clarification, jokes, or Brief table talk, dice resolution, rules clarification, jokes, or
administrative comments do not make a gameplay scene meta. A short recollection administrative comments do not make a gameplay scene meta. A short recollection

View File

@@ -27,6 +27,8 @@ messages:
content_file: ./sharedassets/common-dnd-transcript-chunk.md content_file: ./sharedassets/common-dnd-transcript-chunk.md
cache_control: cache_control:
type: ephemeral type: ephemeral
- role: user
content_file: ./sharedassets/common-dnd-scene-combat-policy.md
- role: user - role: user
content_file: ./instructions.md content_file: ./instructions.md
cache_control: cache_control:

View File

@@ -6,7 +6,8 @@
"required": ["kind", "title", "summary"], "required": ["kind", "title", "summary"],
"properties": { "properties": {
"kind": { "kind": {
"type": "string" "type": "string",
"enum": ["combat", "narrative", "recap", "meta"]
}, },
"title": { "title": {
"type": "string" "type": "string"

View File

@@ -0,0 +1,8 @@
Classify only whether substantive active combat occurs in the supplied
transcript chunk under the shared combat policy. Do not judge the scene title,
summary, non-combat subtype, scene boundary, or any other aspect of a scene
description.
Return `combat` when the chunk contains substantive active combat and
`non_combat` otherwise. Give a concise, transcript-grounded explanation for
the classification.

View File

@@ -0,0 +1,23 @@
id: dnd.scene_descriptions.validate_combat
version: "v1"
default_profile: dnd-extraction
inputs:
- name: transcript
required: true
content_type: application/json
messages:
- role: system
content_file: ./sharedassets/common-dnd-system.md
- role: user
content_file: ./sharedassets/common-dnd-scene-combat-policy.md
- role: user
content_file: ./instructions.md
- role: user
content_file: ./sharedassets/common-dnd-transcript-chunk.md
cache_control:
type: ephemeral
output:
format: json
validation_mode: json_schema
schema_path: dnd_scene_combat_semantics_llm.v1.json
repair_attempts: 1

View File

@@ -0,0 +1,18 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "notarius.dnd.scene_descriptions.combat_semantics.llm",
"type": "object",
"additionalProperties": false,
"required": ["classification", "explanation"],
"properties": {
"classification": {
"type": "string",
"enum": ["combat", "non_combat"]
},
"explanation": {
"type": "string",
"minLength": 1,
"maxLength": 512
}
}
}

View File

@@ -1,6 +1,8 @@
Transcript units are the only evidence for extracted events and factual claims. Transcript units are the only evidence for extracted events and factual claims.
Every reported factual claim must be supported by cited transcript units. Use Every reported factual claim must be supported by cited transcript units. Use
integer `start_unit_id` and `end_unit_id` values from the transcript. integer `start_unit_id` and `end_unit_id` values from the transcript.
Within each range, `start_unit_id` must identify the earlier transcript unit and
`end_unit_id` the same or a later unit according to transcript order.
When supporting evidence is non-contiguous, use multiple narrow ranges rather When supporting evidence is non-contiguous, use multiple narrow ranges rather
than a broad range that bridges unrelated conversation. than a broad range that bridges unrelated conversation.

View File

@@ -0,0 +1,13 @@
Use `combat` only when substantive active combat materially organizes the
scene. Active combat includes initiative or turn exchanges, attacks, combat
spells, damage, saves, movement, or similarly sustained hostile action.
Do not use `combat` for planning or preparing for a possible fight; threats,
hostile dialogue, or a tense confrontation; immediate aftermath, looting,
healing, or discussion of a completed fight; a recap or in-world recollection
of earlier combat; or out-of-character rules discussion without active
encounter play.
When a chunk contains substantive active combat alongside brief setup, rules
clarification, interruption, phase transition, or immediate aftermath, classify
it as `combat`.

View File

@@ -1,3 +1,3 @@
Candidate material: Candidate material, including the exact valid candidate-number range:
{{ input "candidates" }} {{ input "candidates" }}

View File

@@ -1,7 +1,9 @@
Use only the positive integer `candidate_id` values supplied in the candidate material. Use only the positive integer `candidate_number` values supplied in the candidate material. Valid candidate numbers are exactly the inclusive `first` through `last` values declared in `candidate_number_range`; use the explicit number attached to each candidate.
Return a duplicate group only when the evidence supports that every selected candidate describes the same underlying entity. Each group must contain at least two distinct candidate IDs, and its `canonical_candidate_id` must be one of those IDs. A candidate may appear in at most one group. Transcript unit `id` values and evidence `start_unit_id` and `end_unit_id` values identify source positions. They are not candidate numbers and must never be used in `candidate_numbers` or `canonical_candidate_number`.
Omit uncertain matches and candidates that should remain distinct. Do not invent candidates or infer an ID from list position. An empty `duplicate_groups` array is valid. Return a duplicate group only when the evidence supports that every selected candidate describes the same underlying entity. Each group must contain at least two distinct candidate numbers, and its `canonical_candidate_number` must be one of those numbers. A candidate may appear in at most one group.
The response must conform exactly to the selected JSON schema. Return IDs only: do not copy candidate names, evidence, transcript text, source identifiers, or source ranges into the response. Omit uncertain matches and candidates that should remain distinct. Do not invent candidates or infer a number that is not explicitly supplied. An empty `duplicate_groups` array is valid.
The response must conform exactly to the selected JSON schema. Return candidate numbers only: do not copy candidate names, evidence, transcript text, source identifiers, or source ranges into the response.

View File

@@ -11,9 +11,9 @@
"items": { "items": {
"type": "object", "type": "object",
"additionalProperties": false, "additionalProperties": false,
"required": ["candidate_ids", "canonical_candidate_id"], "required": ["candidate_numbers", "canonical_candidate_number"],
"properties": { "properties": {
"candidate_ids": { "candidate_numbers": {
"type": "array", "type": "array",
"minItems": 2, "minItems": 2,
"items": { "items": {
@@ -21,7 +21,7 @@
"minimum": 1 "minimum": 1
} }
}, },
"canonical_candidate_id": { "canonical_candidate_number": {
"type": "integer", "type": "integer",
"minimum": 1 "minimum": 1
} }

View File

@@ -0,0 +1,76 @@
# ADR-0016: Use feedback-aware module-requested retries
**Status:** Accepted
**Date:** 2026-08-28
## Context
An LLM-backed module can discover that a structurally valid model response is
unsafe while translating it into a typed candidate, before the ordinary
validator chain runs. Semantic registry reconciliation is the first such case:
the shared core can identify invalid duplicate-group proposals, and typed
application can reject a domain-incompatible group while preserving a safe
partial result. Repeating the original request without the rejected response or
corrective explanation gives the model no information with which to improve.
The existing feedback-aware validation mechanism already establishes the safe
correction protocol, but module-owned retry directives also carry internal
reason codes, operator messages, and fallback diagnostics. Those values are not
model instructions. Some module retry conditions, including exhausted
structured-output repair, also have no valid exact response to append.
## Decision
An LLM-backed normalizer may attach optional, bounded model-facing correction
guidance to a module-requested retry. Guidance is a separate contract field; the
framework never derives it from a reason code, operator message, diagnostic, or
error. A feedback-capable directive must include the exact model candidate that
controlled the safe fallback and must use `single_response_v1`.
The artifact-neutral producer-attempt state machine constructs the correction
from that exact latest response and the supplied guidance. The producer rebuilds
its complete ordinary request and appends the response as one assistant message
and the guidance as one user message. Earlier attempts do not accumulate, and
the attempt consumes the same configured stage retry budget as operational,
structural, validation, and feedback-free module retries.
A module retry without guidance remains valid and starts a fresh attempt. This
is the required behavior when no valid model candidate exists, including after
structured-output repair is exhausted. If feedback is supplied without a valid
supported candidate, the framework fails the module contract rather than
retrying blindly or inventing guidance.
After retry exhaustion, the normalizer's safe fallback continues through its
validator chain. Safe semantic groups may remain applied, unsafe groups remain
separate, and bounded fallback diagnostics may explain the process degradation.
Exact responses and correction text remain attempt-local and are excluded from
ordinary errors, warnings, manifests, receipts, caches, and checkpoints.
This decision extends, rather than supersedes,
[ADR-0014](0014-feedback-aware-validation-retries.md): both correction sources
use the same transport-neutral payload, replacement-request semantics, outer
retry budget, and sensitive-data boundary.
## Alternatives considered
- Continue blind module retries. This preserves a smaller contract but wastes
the module's deterministic diagnosis and commonly repeats the same defect.
- Convert module safety checks into validators. Typed reconciliation must apply
only safe proposal groups and retain a fallback before validation; moving
artifact-owned translation and application policy into validators would blur
stage ownership.
- Copy the retry reason or operator message into the model request. Those values
are written for provenance and humans, can contain opaque internal labels,
and do not reliably describe a correct replacement.
- Require feedback for every module retry. Structural failures may have no
valid exact candidate, so this would either prevent useful fresh retries or
fabricate prior-response material.
## Consequences
The normalize retry contract and generic producer-attempt directive gain an
optional correction-guidance field and candidate-pairing validation. Modules
that use it must provide semantically meaningful bounded prose and an exact
candidate. Registry reconciliation maintains separate operator and model
renderers, and policy fingerprints change so checkpoints created under blind
retry behavior are not reused.

View File

@@ -55,8 +55,8 @@ pipeline ID and **--input** are required.
| **--session-id id** | Override the generated prompt session identifier with a non-empty value for LLM-backed module calls. | | **--session-id id** | Override the generated prompt session identifier with a non-empty value for LLM-backed module calls. |
| **--reasoning-effort value** | Replace the selected PromptKit profile's reasoning effort for every LLM-backed call in this run. The value must be non-empty and the flag may be specified only once. | | **--reasoning-effort value** | Replace the selected PromptKit profile's reasoning effort for every LLM-backed call in this run. The value must be non-empty and the flag may be specified only once. |
| **--clear-reasoning-effort** | Clear reasoning effort inherited from the selected PromptKit profile for every LLM-backed call in this run. | | **--clear-reasoning-effort** | Clear reasoning effort inherited from the selected PromptKit profile for every LLM-backed call in this run. |
| **--reference selector=path** | Add or replace a file reference binding. Repeatable. | | **--reference selector=path** | Add or replace external file reference bindings at pipeline, lane, chunk, or binding scope. Repeatable. |
| **--without-reference selector** | Remove a configured optional reference binding. Repeatable. | | **--without-reference selector** | Remove matching configured external reference bindings. Repeatable. |
**--chunk_cache** accepts only **auto**, **bypass**, or **refresh**. **--chunk_cache** accepts only **auto**, **bypass**, or **refresh**.
**--debug-dir**, **--output-dir**, **--session-id**, and **--debug-dir**, **--output-dir**, **--session-id**, and
@@ -81,23 +81,49 @@ guidance.
### Reference selectors ### Reference selectors
Use **--reference** only for a reference slot declared by the selected Use **--reference** only for reference slots declared by the selected
configured target. The accepted selector forms are: configured targets. Qualification narrows the scope of an override:
| Form | Target | | Form | Target |
| --- | --- | | --- | --- |
| slot=path | The unique selected target that declares slot. | | slot=path | Every selected target that declares slot. |
| chunk.slot=path | The chunker. | | chunk.slot=path | The chunker. |
| merge.slot=path | The unique selected merger that declares slot. | | lane.slot=path | Every extractor, merger, or normalizer in lane that declares slot. |
| lane.slot=path | The unique extractor, merger, or normalizer in lane that declares slot. |
| lane.extract.slot=path | The extractor in lane. | | lane.extract.slot=path | The extractor in lane. |
| lane.merge.slot=path | The merger in lane. | | lane.merge.slot=path | The merger in lane. |
| lane.normalize.slot=path | The normalizer in lane. | | lane.normalize.slot=path | The normalizer in lane. |
**--without-reference** uses the same selector forms without =path. Slot Pipeline- and lane-scoped selectors are expected to match multiple targets and
names, requiredness, and configured bindings are part of the fail if they match none. A stage-specific selector fails when its lane is not
selected or its target does not declare the slot. There is no stage-wide
`merge.slot` shorthand; name the lane when targeting a merger.
CLI bindings override configured external paths. For overlapping CLI
selectors, a binding-specific or chunk selector overrides a lane selector, and
a lane selector overrides a pipeline selector. The last occurrence wins at
equal scope. Binding and unbinding the same concrete target at equal scope is
an error; a narrower bind or unbind may create an intentional exception to a
broader action.
**--without-reference** uses the same selector forms without `=path` and
removes external bindings only. Neither flag replaces or removes a generated
artifact handoff; an external/generated collision is a resolution error.
Required slots are checked after all effective changes. CLI reference paths
are resolved relative to the process working directory, so subprocess and
service callers should use absolute paths. Slot names, accepted media types,
size limits, requiredness, and configured generated bindings are part of the
[configuration contract](config.md). [configuration contract](config.md).
For example, one shared campaign reference can reach every compatible target,
with an optional lane-specific exception:
~~~
notarius run dnd-session \
--input /data/transcript.json \
--reference party=/data/references/party.txt \
--reference npc-registry.party=/data/references/npc-party-context.txt
~~~
### Run output ### Run output
Without **--json**, standard output contains the completed pipeline ID, counts Without **--json**, standard output contains the completed pipeline ID, counts

View File

@@ -411,10 +411,15 @@ slot. A generated binding supplies one accepted normalized artifact; it does
not name a file. A configured generated dependency remains required even when not name a file. A configured generated dependency remains required even when
that consumer slot is otherwise optional. that consumer slot is otherwise optional.
Pipeline references are defaults. A matching step-local or binding-local Pipeline references are configuration defaults. A matching step-local or
external path overrides a pipeline default. Required slots must be bound after binding-local external path overrides a pipeline default. CLI reference
these configuration values and any CLI reference overrides are applied. bindings are then operational overrides of configured external paths; their
Reference paths in YAML are resolved relative to the configuration file. pipeline, lane, and binding scopes and precedence are defined by the
[CLI reference](cli.md#reference-selectors). A CLI file reference cannot
replace a configured generated artifact handoff. Required slots must be bound
after configuration and CLI reference actions are applied. Reference paths in
YAML are resolved relative to the configuration file; CLI reference paths are
resolved relative to the process working directory.
### D&D Reference Slots ### D&D Reference Slots
@@ -504,11 +509,33 @@ Available validator keys are:
| Item occurrences | **extract/dnd/item-occurrences/shape**, **extract/dnd/item-occurrences/registry**, **extract/dnd/item-occurrences/source_refs**, **extract/dnd/item-occurrences/source_relatedness**, **normalize/dnd/item-occurrences/invariants** | | Item occurrences | **extract/dnd/item-occurrences/shape**, **extract/dnd/item-occurrences/registry**, **extract/dnd/item-occurrences/source_refs**, **extract/dnd/item-occurrences/source_relatedness**, **normalize/dnd/item-occurrences/invariants** |
| Item registry | **extract/dnd/item-registry/shape**, **extract/dnd/item-registry/source_refs**, **extract/dnd/item-registry/source_relatedness**, **normalize/dnd/item-registry/identity** | | Item registry | **extract/dnd/item-registry/shape**, **extract/dnd/item-registry/source_refs**, **extract/dnd/item-registry/source_relatedness**, **normalize/dnd/item-registry/identity** |
| NPC occurrences | **extract/dnd/npc-occurrences/shape**, **extract/dnd/npc-occurrences/registry**, **extract/dnd/npc-occurrences/source_refs**, **extract/dnd/npc-occurrences/source_relatedness**, **normalize/dnd/npc-occurrences/invariants** | | NPC occurrences | **extract/dnd/npc-occurrences/shape**, **extract/dnd/npc-occurrences/registry**, **extract/dnd/npc-occurrences/source_refs**, **extract/dnd/npc-occurrences/source_relatedness**, **normalize/dnd/npc-occurrences/invariants** |
| Scene descriptions | **extract/dnd/scene-descriptions/shape**, **extract/dnd/scene-descriptions/source_refs**, **extract/dnd/scene-descriptions/source_relatedness**, **normalize/dnd/scene-descriptions/invariants** | | Scene descriptions | **extract/dnd/scene-descriptions/shape**, **extract/dnd/scene-descriptions/source_refs**, **extract/dnd/scene-descriptions/source_relatedness**, **extract/dnd/scene-descriptions/combat_semantics** (LLM-backed, opt-in), **normalize/dnd/scene-descriptions/invariants** |
| Enemy events | **extract/dnd/enemy-events/shape**, **extract/dnd/enemy-events/engagements**, **extract/dnd/enemy-events/source_refs**, **extract/dnd/enemy-events/source_relatedness**, **normalize/dnd/enemy-events/invariants** | | Enemy events | **extract/dnd/enemy-events/shape**, **extract/dnd/enemy-events/engagements**, **extract/dnd/enemy-events/source_refs**, **extract/dnd/enemy-events/source_relatedness**, **normalize/dnd/enemy-events/invariants** |
| Location registry | **extract/dnd/location-registry/shape**, **extract/dnd/location-registry/source_refs**, **extract/dnd/location-registry/source_relatedness**, **normalize/dnd/location-registry/identity** | | Location registry | **extract/dnd/location-registry/shape**, **extract/dnd/location-registry/source_refs**, **extract/dnd/location-registry/source_relatedness**, **normalize/dnd/location-registry/identity** |
| Location occurrences | **extract/dnd/location-occurrences/shape**, **extract/dnd/location-occurrences/registry**, **extract/dnd/location-occurrences/source_refs**, **extract/dnd/location-occurrences/source_relatedness**, **normalize/dnd/location-occurrences/invariants** | | Location occurrences | **extract/dnd/location-occurrences/shape**, **extract/dnd/location-occurrences/registry**, **extract/dnd/location-occurrences/source_refs**, **extract/dnd/location-occurrences/source_relatedness**, **normalize/dnd/location-occurrences/invariants** |
`extract/dnd/scene-descriptions/combat_semantics` is not in a production default chain. To opt in, replace the scene extractor validator chain with the current ordered chain plus the semantic validator last, and set a positive producer retry budget if a rejection should request a corrected scene:
~~~yaml
extract:
module: dnd/scene-descriptions
retries: 1
validators:
- generic/valid_json
- extract/dnd/scene-descriptions/shape
- extract/dnd/scene-descriptions/source_refs
- generic/valid_json_schema
- extract/dnd/scene-descriptions/source_relatedness
- module: extract/dnd/scene-descriptions/combat_semantics
retries: 1
~~~
An override replaces, rather than extends, the default chain. See [Module Bindings And Validators](#module-bindings-and-validators) for binding, profile, repair, retry, and failure-policy rules.
The validator retry shown above permits one additional execution against the
same scene candidate when the LLM-backed validator itself fails; it is separate
from both the extractor's producer retry and PromptKit structural repair.
When no override is configured, production D&D bindings use the following When no override is configured, production D&D bindings use the following
ordered chains. Each row lists extract then normalize; spell chains are the ordered chains. Each row lists extract then normalize; spell chains are the
same at both stages. same at both stages.

View File

@@ -24,7 +24,9 @@ files. Use absolute paths for service and subprocess deployments. In
particular, observe these different resolution rules: particular, observe these different resolution rules:
- reference paths in YAML are resolved relative to the Notarius configuration - reference paths in YAML are resolved relative to the Notarius configuration
file; and file;
- reference paths passed with `--reference` are resolved relative to the
Notarius process working directory; and
- `promptkit.profile_file` is resolved relative to the Notarius process working - `promptkit.profile_file` is resolved relative to the Notarius process working
directory. directory.
@@ -73,9 +75,21 @@ notarius run dnd-session \
--config /absolute/path/to/notarius.yml \ --config /absolute/path/to/notarius.yml \
--input /absolute/path/to/transcripts/final.trimmed.json \ --input /absolute/path/to/transcripts/final.trimmed.json \
--output-dir /absolute/path/to/notarius-output \ --output-dir /absolute/path/to/notarius-output \
--reference party=/absolute/path/to/references/party.txt \
--reference players=/absolute/path/to/references/players.txt \
--reference glossary=/absolute/path/to/references/glossary.txt \
--reference spell_catalog=/absolute/path/to/references/spells.json \
--json --json
``` ```
Each unqualified reference is pipeline-scoped: Notarius supplies it to every
selected D&D target that declares the slot. A deployment may omit an optional
reference it does not maintain, and may use the lane- or binding-qualified
forms from the [CLI reference](../cli.md#reference-selectors) for an exceptional
override. The registry, scene-description, combat-turn, and NPC-occurrence
references declared between ordered steps in the complete configuration are
generated artifacts. Do not pass those handoffs on the CLI.
The caller should: The caller should:
- capture stdout and stderr separately; - capture stdout and stderr separately;

View File

@@ -31,13 +31,36 @@ notarius run pipeline-id \
``` ```
Use absolute paths for supplied input, configuration, output-root, and Use absolute paths for supplied input, configuration, output-root, and
reference files. Notarius generates a stable prompt session for the resolved reference files. Pass each external reference as its own argument-vector pair;
input module and exact input bytes. Pass **--session-id** only when intentionally do not construct and invoke a shell command. An unqualified reference selector
grouping different invocations under a different session. Supply credentials supplies that file to every compatible selected target. Lane and stage
through Notarius's documented configuration and environment mechanisms, never qualification are available for exceptional overrides, while generated
as command-line arguments or generated secret-bearing configuration. In same-run references remain part of configured pipeline composition. The
particular, a session identifier is provider-visible and is not a credential [CLI reference](../cli.md#reference-selectors) owns the exact selector and
mechanism. precedence contract.
The maintained D&D subprocess workflow uses this facility for campaign context:
```sh
notarius run dnd-session \
--config /absolute/path/to/notarius.yml \
--input /absolute/path/to/transcripts/final.trimmed.json \
--output-dir /absolute/path/to/notarius-output \
--reference party=/absolute/path/to/references/party.txt \
--reference players=/absolute/path/to/references/players.txt \
--reference glossary=/absolute/path/to/references/glossary.txt \
--reference spell_catalog=/absolute/path/to/references/spells.json \
--json
```
Only pass the external references available to and desired by the deployment.
Notarius generates a stable prompt session for the resolved input module and
exact input bytes; reference changes do not change it. Pass **--session-id**
only when intentionally grouping different invocations under a different
session. Supply credentials through Notarius's documented configuration and
environment mechanisms, never as command-line arguments or generated
secret-bearing configuration. In particular, a session identifier is
provider-visible and is not a credential mechanism.
Wait for the process before interpreting standard output. Only an exit status Wait for the process before interpreting standard output. Only an exit status
of 0 permits decoding the receipt. On a nonzero exit, retain standard error for of 0 permits decoding the receipt. On a nonzero exit, retain standard error for

View File

@@ -115,6 +115,16 @@ to checkpoint identity and `pipeline.RunInput`. The public flag and stability
contract are defined by the [CLI reference](../cli.md#run); framework and LLM contract are defined by the [CLI reference](../cli.md#run); framework and LLM
packages only transport the supplied value. packages only transport the supplied value.
The CLI also owns the scope grammar for reference flags. It enumerates the
selected chunk and lane targets from registered module specifications, expands
pipeline- and lane-scoped actions into exact stage-and-lane bindings, and
resolves overlapping bind and unbind actions by specificity before calling
configuration resolution. The generic pipeline therefore receives only exact
`ReferenceBinding` and `ReferenceUnbind` values and has no knowledge of CLI
selector syntax. Configuration resolution retains ownership of configured
external/generated conflicts, required slots, and module compatibility; file
materialization still occurs afterward.
For `run --json`, the CLI constructs and encodes its private run-result receipt For `run --json`, the CLI constructs and encodes its private run-result receipt
after a successful runner result is available, before it publishes logical after a successful runner result is available, before it publishes logical
output files. It writes the prepared receipt to standard output only after output files. It writes the prepared receipt to standard output only after
@@ -166,8 +176,9 @@ is discoverable.
- **internal/cli/production_contract_test.go** covers registrar composition, - **internal/cli/production_contract_test.go** covers registrar composition,
production catalog contents, assets, and representative configuration production catalog contents, assets, and representative configuration
validation. validation.
- **internal/cli/reference_contract_test.go** covers CLI reference overrides, - **internal/cli/reference_contract_test.go** covers scoped CLI reference
origin separation, and materialization boundaries. expansion, specificity, bind/unbind conflicts, generated-reference
protection, origin separation, and materialization boundaries.
- **internal/cli/state_hardening_test.go** covers safe run identity, state - **internal/cli/state_hardening_test.go** covers safe run identity, state
roots, and failure ordering. roots, and failure ordering.

View File

@@ -29,13 +29,17 @@ The D&D registrar registers the familys artifact codecs, extractors, typed
append-order mergers, normalizers, validators, prompt assets, fallback LLM append-order mergers, normalizers, validators, prompt assets, fallback LLM
profile asset, and default validator chains. Each extractor and normalizer has profile asset, and default validator chains. Each extractor and normalizer has
a stable module spec, explicit execution class, strict option decoding, and a a stable module spec, explicit execution class, strict option decoding, and a
typed builder. Scene chunking, every extractor, and NPC, location, and item-registry typed builder. Scene chunking, every extractor, and NPC, location, and
normalization are registered as `llm_backed`; the remaining current D&D mergers item-registry normalization are registered as `llm_backed`; the remaining
and normalizers are `deterministic`. The metadata is available to catalog inspection and current D&D mergers and normalizers are `deterministic`. The metadata is
resolved-pipeline debug data and determines which selected bindings inherit the available to catalog inspection and resolved-pipeline debug data and determines
pipeline profile. The registry normalizers use `single_response_v1`, forwarding which selected bindings inherit the pipeline profile. The registry normalizers
corrections to their reconciliation completion and retaining the accepted raw use `single_response_v1`, forwarding corrections to their reconciliation
proposal only as an owned model candidate. Configuration remains the canonical owner of the exact keys, completion and retaining the accepted raw proposal only as an owned model
candidate. When deterministic proposal safety or typed application rejects a
group, they provide separate model-facing prose for a corrective module retry;
internal issue categories, reason codes, and operator messages remain
diagnostic-only. Configuration remains the canonical owner of the exact keys,
profile precedence, and validator order. profile precedence, and validator order.
Private structured-LLM response schemas are deliberately minimal. They reject Private structured-LLM response schemas are deliberately minimal. They reject
@@ -117,14 +121,27 @@ presentation, and final ephemeral generic transcript windows. These orders and
cache controls are prompt behavior; change them only through the owning cache controls are prompt behavior; change them only through the owning
manifest and prompt declaration. manifest and prompt declaration.
NPC, item, and location registry reconciliation translate shared proposal
safety categories into bounded prose that references only the response-local
duplicate-group ordinals and candidate handles. Item reconciliation appends its
typed rule that currency may be consolidated only with aliases of the same
denomination and never with non-currency items. The next normalize attempt
receives that prose with the exact defective proposal under the shared
replacement-request protocol. Structurally invalid output has no valid proposal
candidate and receives a fresh feedback-free attempt instead. If the stage
budget is exhausted, safe groups stay applied, unsafe groups stay separate, and
one fallback warning summarizes the final defect without raw model content.
## Evidence, Candidates, And Normalization ## Evidence, Candidates, And Normalization
The current transcript is the only durable evidence source. Extractors assign The current transcript is the only durable evidence source. Extractors assign
the current source identity, preserve candidate evidence ranges for validators, the current source identity and losslessly order any reversed range whose two
and canonically order or remove exact duplicate ranges without asking the endpoints resolve in that source, using transcript position rather than numeric
model to repair semantic errors. Campaign context and generated artifacts may unit-ID order. They then canonically order ranges and remove exact duplicates.
ground names or control routing, but they never establish evidence for a D&D This routine canonicalization does not request a retry or emit a warning.
result. Unresolvable or wrong-source ranges remain unchanged for validators to reject.
Campaign context and generated artifacts may ground names or control routing,
but they never establish evidence for a D&D result.
Default chains keep responsibilities separate: structural validators assess the Default chains keep responsibilities separate: structural validators assess the
candidate, source-reference validators resolve cited ranges against the current candidate, source-reference validators resolve cited ranges against the current
@@ -134,18 +151,115 @@ relatedness validators report advisory evidence concerns. The configured order
is documented in is documented in
[Configuration](../config.md#production-validator-keys-and-default-chains). [Configuration](../config.md#production-validator-keys-and-default-chains).
Every D&D rejection describes the correction in transcript-grounded domain The optional `extract/dnd/scene-descriptions/combat_semantics` validator is the
terms, using contextual names, artifact fields, and source segment ranges when D&D family's LLM-backed review of only combat versus non-combat classification.
useful. The guidance must not ask the model to reproduce durable entity IDs, It selects the shared combat-policy prompt fragment and asks the model to
hashes, validator module keys, or reason codes. Those identifiers remain in classify the current chunk independently as `combat` or `non_combat` without
ordinary validation provenance; only the actionable semantic guidance is receiving the proposed scene kind. Deterministic code compares that
eligible for the correction prompt. classification with the proposed kind and either approves it or produces the
appropriate correction guidance. This keeps every schema-valid classification
interpretable and avoids anchoring the reviewer on the producer's answer. It
does not assess titles, summaries, non-combat subtype, or scene boundaries;
deferred boundary-coherence review remains separate. It is opt-in;
[Configuration](../config.md) owns selection and retry/failure behavior.
### Combat-semantics provider evaluation
The human-reviewed corpus at
`internal/modules/dnd/validate/scenedescriptions/combat_semantics/testdata/evaluation_cases.json`
owns the proposed kind, expected combat classification, and reviewer rationale
for each synthetic case. Its package test validates the fixture contract only.
Provider evaluation remains an explicit maintainer operation and must not be
added to the default offline test suite.
Use the following protocol before proposing default-chain inclusion:
1. Record the Notarius commit, prompt and schema fingerprints, provider, model,
profile settings, reasoning effort, structural-repair setting, number of
repetitions, and evaluation date before collecting results. Do not revise
expected classifications merely to agree with provider output; a substantive
corpus correction requires independent human review.
2. Exercise the production validator construction and prompt assets from an
explicitly invoked, disposable evaluation driver or test in the validator
package. For each corpus case, construct transcript source units from the
listed IDs and text, assign matching per-unit source references, and use
`source.MaterializeChunkPlan` with one range spanning those units. Construct
exactly one scene whose ID and source range match that chunk and whose kind
is the case's `proposed_kind`; title and summary may use fixed placeholders
because the validator neither receives nor evaluates them. Invoke the typed
validator at extract stage through the production registry and scheduled LLM
client. Do not commit provider credentials, generated source material, or an
always-on live test.
3. Compare the model classification with `expected_classification`, then verify
that its deterministic comparison with `proposed_kind` yields the expected
approval or rejection direction.
Record an unexpected approval of an expected rejection as a false
acceptance, an unexpected rejection of an expected approval as a false
rejection, and any validator execution failure separately from semantic
accuracy. Retain per-case results so repeated trials and systematic failure
modes remain visible.
4. Evaluate producer correction separately with representative complete
scene-description runs configured as shown in
[Configuration](../config.md#production-validator-keys-and-default-chains).
For every initial semantic rejection, record whether the next producer
attempt returns the requested combat status and is approved. Do not count a
PromptKit structural repair as a producer-correction attempt.
5. Run the correction evaluation with debug capture enabled and without reused
extraction checkpoints. Record added validator and producer calls, elapsed
latency, and cumulative prompt, completion, cached, and total token usage
from the debug attempt records. Compare these values with an otherwise
identical run whose scene-description chain omits the semantic validator.
The default-chain review must consider classification error, false acceptance,
false rejection, execution failure, producer-correction success, added calls,
latency, and token use together. A structurally successful provider run alone
is not evidence that the validator should become a default.
Every producer-correctable D&D rejection describes all currently detectable
corrections in transcript-grounded domain terms, using contextual names,
model-owned artifact fields, and source segment ranges when useful. Validators
collect independent record defects in one pass so one retry does not merely
reveal the next issue. Shared D&D diagnostic helpers keep repeated rules and
record descriptions stable, de-duplicated, and bounded; each artifact family
continues to own the semantic rule and its prose.
Operator diagnostics and model guidance are separate products of the same
assessment. Operator messages may use typed paths, reason details, and opaque
application identities. Correction guidance must not copy those messages or
ask the model to reproduce durable entity IDs, hashes, validator module keys,
reason codes, or Go field paths. A registry-normalization rejection instead
speaks in terms of the duplicate-group proposal response the normalizer can
actually revise. Normalization-only deterministic invariants retain useful
operator detail but do not imply that a model controls derived ordering or
identity. Only bounded actionable semantic guidance is eligible for a
correction prompt.
Private LLM schemas use simple enums for closed categorical fields when the
provider-compatible shape can express the rule directly. Deterministic typed
validators retain the same checks as defense in depth and for non-LLM
producers. Private schemas keep every property required and avoid optional
properties, `uniqueItems`, and conditional cross-field logic.
Item-occurrence shape validation groups repeated holder mistakes by occurrence
kind and gives the producer the required JSON null/non-null relationship. It
identifies affected records by contextual item name and cited transcript range,
never by the deterministically attached durable item ID. Holder mistakes remain
semantic rejections rather than silent rewrites because changing a holder can
also change the meaning of the occurrence kind.
Enemy-event extraction additionally rejects a second `engaged` observation for Enemy-event extraction additionally rejects a second `engaged` observation for
the same comparison identity within one scene-scoped result. Normalization may the same comparison identity within one scene-scoped result. Normalization may
combine results from distinct scenes, so it intentionally does not apply that combine results from distinct scenes, so it intentionally does not apply that
rule. Configuration owns the exact validator key and chain position. rule. Configuration owns the exact validator key and chain position.
Extraction source-reference validators share one full-span chunk-containment
policy. After ordinary reference validity succeeds, the policy resolves both
endpoints through document order and requires every source unit in the
inclusive range to be present in the current chunk. It does not assume numeric
unit-ID ordering, mutate input, or weaken wrong-source and unresolved-reference
validation. Scene descriptions remain separate because their validator owns an
exact one-scene range contract rather than general extraction containment.
Normalizers are deterministic for spells, combat turns, item occurrences, NPC Normalizers are deterministic for spells, combat turns, item occurrences, NPC
occurrences, scene descriptions, enemy events, and location occurrences. They occurrences, scene descriptions, enemy events, and location occurrences. They
canonicalize display values and evidence, use source-document order for stable canonicalize display values and evidence, use source-document order for stable
@@ -158,13 +272,16 @@ proposal to reconcile identity groups.
The three registry normalizers instantiate the domain-neutral The three registry normalizers instantiate the domain-neutral
`internal/framework/semanticreconcile` engine with default bounds. Each `internal/framework/semanticreconcile` engine with default bounds. Each
eligible candidate receives a contiguous, one-based `candidate_id` for that eligible candidate receives a contiguous, one-based `candidate_number` for that
request. The model sees that handle, the candidate label and source-free request, and candidate material declares the exact inclusive range. The model
evidence ranges, plus bounded transcript windows; it returns only duplicate sees that handle, the candidate label and source-free evidence ranges, plus
bounded transcript windows; it returns only duplicate
groups of supplied handles and one supplied canonical handle per group. It groups of supplied handles and one supplied canonical handle per group. It
never returns names, evidence, durable IDs, or replacement records. Identical never returns names, evidence, durable IDs, or replacement records. Identical
labels and evidence remain independently selectable because their handles are labels and evidence remain independently selectable because their handles are
distinct. distinct. Transcript unit `id` values and evidence `start_unit_id` and
`end_unit_id` values are source positions in a separate namespace and are
never valid candidate numbers.
The generic core owns the mandatory handle protocol, candidate and transcript The generic core owns the mandatory handle protocol, candidate and transcript
presentation, the private response schema, source-reference validation, presentation, the private response schema, source-reference validation,

View File

@@ -247,8 +247,12 @@ redaction boundary.
Structural repair does not replace pipeline retry behavior: a binding's Structural repair does not replace pipeline retry behavior: a binding's
configured retry count reruns its complete stage attempt after an operational configured retry count reruns its complete stage attempt after an operational
or structural error, module-requested retry, or actionable semantic rejection. or structural error, module-requested retry, or actionable semantic rejection.
The pipeline owns attempt lifecycle, validation chains, and retry diagnostics; An actionable module-requested retry and a validator rejection both use the
see [Pipeline Internals](pipeline.md#validation-retries-and-output) and the same correction payload when the producer exposes an exact latest response;
feedback-free module retries reconstruct the ordinary request without appended
messages. The pipeline owns attempt lifecycle, validation chains, and retry
diagnostics; see
[Pipeline Internals](pipeline.md#validation-retries-and-output) and the
[binding reference](../config.md#module-bindings-and-validators). [binding reference](../config.md#module-bindings-and-validators).
## Timeout Ownership ## Timeout Ownership

View File

@@ -95,19 +95,24 @@ source-backed artifact-family normalizer projects its deterministic records
into contextual candidates and owned typed record envelopes, supplies its into contextual candidates and owned typed record envelopes, supplies its
chosen prompt identity and resolved LLM profile, and constructs an engine with chosen prompt identity and resolved LLM profile, and constructs an engine with
explicit limits. The core filters invalid evidence, assigns contiguous explicit limits. The core filters invalid evidence, assigns contiguous
request-local integer handles, renders bounded candidate and transcript one-based request-local candidate numbers, renders bounded candidate and
materials, invokes the structured-completion boundary, and assesses the transcript materials, invokes the structured-completion boundary, and assesses
returned duplicate groups into a stable non-overlapping plan. the returned duplicate groups into a stable non-overlapping plan. Candidate
material declares the exact inclusive number range for the request. Transcript
unit IDs and evidence range endpoints remain source positions in a separate
namespace and are never valid candidate numbers.
The normalizer then applies that plan through a typed `ApplicationPolicy`. The The normalizer then applies that plan through a typed `ApplicationPolicy`. The
core preserves ungrouped records, contribution order, and provenance while the core preserves ungrouped records, contribution order, and provenance while the
artifact family owns group guards, field and evidence consolidation, durable artifact family owns group guards, field and evidence consolidation, durable
ID derivation, retry and fallback presentation, classified diagnostics, and postconditions. ID derivation, retry and fallback presentation, classified diagnostics, and
Request-local handles do not enter the typed value or durable artifact. Fewer postconditions.
than two eligible candidates skips model invocation; exceeding a candidate or These candidate numbers are the concrete private representation of ADR-0013's
combined-material bound preserves the deterministic result under the family's request-local handles; they do not enter the typed value or durable artifact.
fallback policy. Provider, transport, cancellation, and context-construction Fewer than two eligible candidates skips model invocation; exceeding a
failures remain execution errors. candidate or combined-material bound preserves the deterministic result under
the family's fallback policy. Provider, transport, cancellation, and
context-construction failures remain execution errors.
When the engine actually makes a proposal call, its typed result carries the When the engine actually makes a proposal call, its typed result carries the
owned exact proposal response under the same correction contract as other owned exact proposal response under the same correction contract as other
@@ -115,6 +120,16 @@ eligible producers. Deterministic skip, limit, and fallback outcomes carry no
model candidate, so a later rejection applies terminal policy without spending model candidate, so a later rejection applies terminal policy without spending
an ineffective semantic retry. an ineffective semantic retry.
When proposal assessment or typed group application rejects a structurally
valid group, the normalizer may return its safe partial value with a
module-requested retry. A feedback-capable directive supplies bounded
model-facing correction guidance separately from operator diagnostics and
retains the exact proposal response as its candidate. The shared stage retry
mechanism appends that response and guidance to a fresh complete request. A
structurally invalid completion has no valid candidate and therefore requests a
feedback-free fresh attempt. On exhaustion, only the final safe fallback and
its bounded process diagnostic advance to validation.
The core supplies a conservative generic prompt and the single private The core supplies a conservative generic prompt and the single private
response schema. A domain prompt may substitute its semantic instructions but response schema. A domain prompt may substitute its semantic instructions but
mounts the core-owned protocol and candidate/transcript presentation assets. mounts the core-owned protocol and candidate/transcript presentation assets.

View File

@@ -97,6 +97,13 @@ codec, checks its complete schema and media identity, and records a content
digest plus bounded producer provenance. A missing, ambiguous, invalid, or digest plus bounded producer provenance. A missing, ambiguous, invalid, or
incompatible producer prevents the consumer step from starting. incompatible producer prevents the consumer step from starting.
Resolution receives only exact stage-and-lane operational reference overrides.
The CLI may offer broader pipeline- or lane-scoped selectors, but expands and
arbitrates those before entering the framework. External overrides are applied
after configured external defaults and local bindings. They cannot coexist
with a generated binding for the same target and slot, and external unbinds do
not remove generated handoffs.
## Execution And Ordering ## Execution And Ordering
The runner validates its input, installs no-op state collaborators when none The runner validates its input, installs no-op state collaborators when none
@@ -154,18 +161,25 @@ complete validation chain. It preserves terminal diagnostics only from the final
or rejected attempt, plus one fixed validation-incomplete warning per validator whose execution or rejected attempt, plus one fixed validation-incomplete warning per validator whose execution
budget was exhausted under `warn_continue`. Cancellation stops retries. budget was exhausted under `warn_continue`. Cancellation stops retries.
Normalizer-specific retry directives consume this same budget and validate any Normalizer-specific retry directives consume this same budget and validate any
final safe fallback through the normalizer chain. final safe fallback through the normalizer chain. A directive may carry bounded
correction guidance only when it also exposes the exact latest
`single_response_v1` candidate. The state machine then uses the same replacement
request shape as validator correction. A directive without guidance clears any
prior correction and starts a fresh attempt, which preserves structural retry
behavior when no valid response exists.
The artifact-neutral producer-attempt state machine owns that shared budget, The artifact-neutral producer-attempt state machine owns that shared budget,
attempt provenance, semantic-correction material, and terminal-policy attempt provenance, semantic-correction material, and terminal-policy
selection. It accepts producer and complete-validation closures, so artifact selection. It accepts producer and complete-validation closures, so artifact
materialization, cache handling, checkpoints, and debug output stay at the materialization, cache handling, checkpoints, and debug output stay at the
operation boundary. It distinguishes operational, structural, module-requested, operation boundary. It distinguishes operational, structural, module-requested,
and semantic retries. A semantic retry is available only for a valid latest and validator-semantic retries. Model feedback from either semantic source is
`single_response_v1` candidate; a deterministic or no-model rejection instead available only for a valid latest `single_response_v1` candidate. A
settles the semantic policy immediately. Structural-output errors alone use the deterministic or no-model validator rejection instead settles the semantic
structural policy, and validation failure without rejection settles the policy immediately, while a feedback-free module directive remains an ordinary
validator-failure policy without regenerating the producer. fresh retry. Structural-output errors alone use the structural policy, and
validation failure without rejection settles the validator-failure policy
without regenerating the producer.
Chunk planning uses this state machine for generated plans. A rejected or Chunk planning uses this state machine for generated plans. A rejected or
validation-incomplete automatic cache hit is not model material and therefore validation-incomplete automatic cache hit is not model material and therefore
@@ -224,6 +238,8 @@ Debug recording is attempt-scoped and application-owned. A failure to persist
required debug data is a framework error. State roots, persistence, reason-code required debug data is a framework error. State roots, persistence, reason-code
meanings, resume, and cleanup are intentionally owned by meanings, resume, and cleanup are intentionally owned by
[Run State Internals](state.md) and [Operations](../operations.md). [Run State Internals](state.md) and [Operations](../operations.md).
Extract-validator trace scopes include the current chunk ordinal so concurrent
chunks cannot overwrite one another's validator attempts or LLM artifacts.
## Invariants To Preserve ## Invariants To Preserve

View File

@@ -201,6 +201,13 @@ they are not model instructions. The framework constructs model-facing retry
text only from the semantic guidance and fails the contract rather than text only from the semantic guidance and fails the contract rather than
inventing or truncating missing guidance. inventing or truncating missing guidance.
An LLM-backed module may also request a feedback-aware retry when its own
deterministic translation or typed safety policy rejects a structurally valid
model response. It must supply model-facing guidance separately from its
reason code, operator message, and fallback diagnostics, together with the
exact `single_response_v1` candidate. A feedback-free module retry remains
valid when no exact candidate exists.
Default validator chains are production composition policy and are registered Default validator chains are production composition policy and are registered
centrally by stage and module. Configuration may replace a stage-local default, centrally by stage and module. Configuration may replace a stage-local default,
including with an explicitly empty chain. Configured validator order is including with an explicitly empty chain. Configured validator order is
@@ -222,12 +229,14 @@ the two budgets must remain separate.
An LLM-backed producer can participate in semantic correction only when it An LLM-backed producer can participate in semantic correction only when it
declares `single_response_v1` and returns the exact one response that directly declares `single_response_v1` and returns the exact one response that directly
controlled its candidate. On an actionable rejection, the framework rebuilds controlled its candidate. On an actionable validator rejection or
the ordinary request and appends only the latest defective response as an feedback-capable module retry, the framework rebuilds the ordinary request and
`assistant` message plus one aggregated `user` correction message. This is a appends only the latest defective response as an `assistant` message plus one
fresh replacement request, not a growing conversation. The retry budgets, aggregated `user` correction message. This is a fresh replacement request, not
terminal policy, and sensitive-data rationale are recorded in a growing conversation. The retry budgets, terminal policy, and sensitive-data
[ADR-0014](../adr/0014-feedback-aware-validation-retries.md). rationale are recorded in
[ADR-0014](../adr/0014-feedback-aware-validation-retries.md) and
[ADR-0016](../adr/0016-feedback-aware-module-requested-retries.md).
When a model selects an application entity, callers must supply a contextual When a model selects an application entity, callers must supply a contextual
selection and deterministically attach the opaque application identity whenever selection and deterministically attach the opaque application identity whenever

84
docs/releases/v0.6.0.md Normal file
View File

@@ -0,0 +1,84 @@
# Notarius v0.6.0
This release improves the reliability and ergonomics of unattended,
subprocess-driven D&D extraction pipelines.
## Summary
Notarius now gives models more precise, semantically useful correction guidance
when deterministic or model-backed validation rejects an otherwise structured
candidate. Semantic reconciliation retries identify candidates with compact,
request-local numbers, preserve valid candidates when a proposal cannot be
repaired, and report bounded process warnings when reconciliation is
incomplete. D&D extraction also canonicalizes safely resolvable reversed source
ranges and applies more consistent schema, prompt, and validator constraints,
improving successful operation with smaller models without weakening evidence
validation.
An optional LLM-backed combat-semantics validator can review whether scene
descriptions correctly consolidate combat and apply the `combat` kind. For
subprocess callers, command-line references now follow the pipeline-scoped
configuration model: a shared reference can be supplied once and automatically
reach every compatible selected target, while lane- and binding-specific forms
remain available for exceptions.
## Compatibility
- The meaning of an unqualified `--reference slot=path` or
`--without-reference slot` selector has changed. It now applies to every
selected pipeline target that declares the slot instead of requiring exactly
one matching target. Review callers that relied on ambiguity rejection or a
unique implicit target; the current selector contract is documented in the
[CLI reference](../cli.md).
- The stage-wide `merge.slot` CLI reference shorthand has been removed. Use
`lane.merge.slot` for an exact merge binding, `lane.slot` for all compatible
bindings in one lane, or an unqualified `slot` for pipeline scope.
- Command-line file references cannot replace, remove, or coexist with a
generated artifact handoff for the same concrete target and slot. Resolution
reports the conflict so the caller can narrow or remove the selector.
- The private LLM-facing semantic-reconciliation contract now uses contiguous
`candidate_number` values rather than application identities. This is not a
durable or operator-configurable contract and requires no operator action.
- Durable D&D artifact schemas, the `notarius.run-result.v2` receipt,
`notarius.warnings.v2`, and `notarius.diagnostics.v1` are unchanged from
`v0.5.0`.
## Upgrade
1. Update subprocess commands to provide shared inputs once with
`--reference slot=path`, and convert any stage-wide `merge.slot` selector to
an explicit supported scope. The [subprocess consumer guide](../consumers/subprocess.md)
and [complete D&D consumer guide](../consumers/dnd-pipeline.md) show the
current invocation pattern.
2. Review CLI reference overrides that overlap generated same-run references;
narrow or remove an external selector rather than attempting to replace the
generated handoff.
3. Run `notarius config validate --config <path> --pipeline <id>` for every
deployed pipeline configuration.
4. Optionally add
`extract/dnd/scene-descriptions/combat_semantics` to the scene-description
validator chain where the deployment wants LLM-backed combat-scene review;
see the [configuration reference](../config.md).
5. Perform a representative run with the deployed model profile and inspect
the machine-readable receipt, warnings, and diagnostics before promotion.
## Changes
- Added feedback-aware, module-requested retry support for normalizers while
keeping retry budgets bounded and preserving attempt diagnostics.
- Reworked semantic reconciliation around compact, contiguous candidate
numbers, explicit namespace guidance, complete proposal validation, and
model-facing retry feedback that omits opaque internal identities.
- Preserved candidates separately when an exhausted semantic proposal cannot
be safely applied, with a bounded process warning describing the fallback.
- Canonicalized safely resolvable reversed D&D source-reference endpoints
before deterministic coverage validation and clarified the shared evidence
prompt.
- Improved D&D schemas, prompts, and validators with closed-value constraints,
contextual correction guidance, shared source-range coverage logic, and
clearer item-holder transition rules.
- Added the optional D&D scene-description combat-semantics validator, shared
combat policy assets, evaluation fixtures, and retry-path coverage.
- Made unqualified CLI references pipeline-scoped, added hierarchical lane and
exact-binding selectors, defined deterministic override precedence, and
documented first-class subprocess use.

View File

@@ -14,47 +14,9 @@ eligibility, and the separation of actionable process warnings from quality
diagnostics. The remaining near-term work applies those completed foundations diagnostics. The remaining near-term work applies those completed foundations
to domain review and empirical evaluation. to domain review and empirical evaluation.
### D&D Combat Scene Semantic Validation Near-term reliability work should now be selected from the concrete evaluation
and extension opportunities below. The retry, validation, and subprocess
- Add an optional production LLM-backed D&D validator that determines whether foundations described above are implemented current behavior.
proposed scene boundaries and classifications represent substantive active
combat correctly. Its central quality goal is that active combat is kept in
coherent scenes classified as `combat`, rather than split incorrectly or
hidden inside scenes classified as `narrative`, `recap`, or `meta`.
- Resolve the validator's exact target before implementation. The current
`dnd/scenes` chunker owns only complete, gap-free source ranges, while the
per-chunk `dnd/scene-descriptions` extractor owns the `combat`, `narrative`,
`recap`, and `meta` classification. The preferred initial placement is
therefore an extract-stage validator for `dnd/scene-descriptions`, where it
can compare one proposed kind with the corresponding transcript chunk.
- Consider a chunk-stage LLM validator only for a distinct boundary-coherence
question that can be answered from the complete transcript and proposed
range map, such as whether one continuous combat was fragmented across
inappropriate scene boundaries. Do not duplicate the same classification
judgment at both stages. Moving classification into chunk-plan annotations
would change the deliberately minimal, annotation-free chunk contract and
requires an explicit architecture review before it is selected.
- Validate both false negatives and false positives: a non-combat kind must not
omit substantive active combat, and a combat kind must be supported by such
combat. Keep the existing deterministic downstream rule that combat-turn
extraction runs only for an exact `combat` scene classification; semantic
review improves the upstream classification but does not replace that gate.
- Run the semantic validator through PromptKit, use a minimal required-field
structured response schema, and let PromptKit repair structural validator
output within its bounded budget. A contract-invalid final validator response
is a validator execution failure, not a semantic rejection and not a reason
to recursively validate the validator.
- Evaluate the prompt and decision policy against a small human-reviewed set
containing combat setup, active turns, interruptions, multi-phase encounters,
brief rules discussion, aftermath, recalled combat, and false-positive
hostile dialogue. Measure false acceptance, false rejection, retry success,
added calls, latency, and token cost before placing it in the production
default chain.
- An ADR is not required if classification remains owned by
`dnd/scene-descriptions` and the validator follows the generic validation ADR.
Create or supersede an ADR if the work transfers scene classification into
the chunker or otherwise changes stage ownership or the durable chunk-plan
contract.
## Near-Term D&D Pipeline ## Near-Term D&D Pipeline

View File

@@ -4,6 +4,7 @@ import (
"bytes" "bytes"
"os" "os"
"path/filepath" "path/filepath"
"slices"
"strings" "strings"
"testing" "testing"
@@ -16,18 +17,14 @@ func TestReferenceSelectorsParseAndApplyAllDocumentedForms(t *testing.T) {
tests := []struct { tests := []struct {
name string name string
selector string selector string
only []string want []string
wantStage pipeline.ModuleStage
wantLane string
wantSlot string
}{ }{
{name: "flat", selector: "alpha-slot", wantStage: pipeline.StageExtract, wantLane: "alpha", wantSlot: "alpha-slot"}, {name: "pipeline", selector: "shared", want: []string{"alpha.extract.shared", "alpha.merge.shared", "alpha.normalize.shared", "beta.extract.shared", "beta.merge.shared", "beta.normalize.shared"}},
{name: "chunk", selector: "chunk.chunk-slot", wantStage: pipeline.StageChunk, wantSlot: "chunk-slot"}, {name: "chunk", selector: "chunk.chunk-slot", want: []string{"chunk.chunk-slot"}},
{name: "merge", selector: "merge.alpha-merge", only: []string{"alpha"}, wantStage: pipeline.StageMerge, wantLane: "alpha", wantSlot: "alpha-merge"}, {name: "lane", selector: "alpha.shared", want: []string{"alpha.extract.shared", "alpha.merge.shared", "alpha.normalize.shared"}},
{name: "lane", selector: "alpha.alpha-slot", wantStage: pipeline.StageExtract, wantLane: "alpha", wantSlot: "alpha-slot"}, {name: "lane extract", selector: "alpha.extract.alpha-slot", want: []string{"alpha.extract.alpha-slot"}},
{name: "lane extract", selector: "alpha.extract.alpha-slot", wantStage: pipeline.StageExtract, wantLane: "alpha", wantSlot: "alpha-slot"}, {name: "lane merge", selector: "alpha.merge.alpha-merge", want: []string{"alpha.merge.alpha-merge"}},
{name: "lane merge", selector: "alpha.merge.alpha-merge", wantStage: pipeline.StageMerge, wantLane: "alpha", wantSlot: "alpha-merge"}, {name: "lane normalize", selector: "alpha.normalize.alpha-normalize", want: []string{"alpha.normalize.alpha-normalize"}},
{name: "lane normalize", selector: "alpha.normalize.alpha-normalize", wantStage: pipeline.StageNormalize, wantLane: "alpha", wantSlot: "alpha-normalize"},
} }
for _, tt := range tests { for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) { t.Run(tt.name, func(t *testing.T) {
@@ -37,70 +34,135 @@ func TestReferenceSelectorsParseAndApplyAllDocumentedForms(t *testing.T) {
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
overrides, _, err := resolveCLIReferenceRequests(cfg, "demo", tt.only, catalog, []cliReferenceRequest{{Selector: selector, Source: "reference.txt"}}, nil) overrides, _, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog, []cliReferenceRequest{{Selector: selector, Source: "reference.txt"}}, nil)
if err != nil { if err != nil {
t.Fatalf("resolve selector: %v", err) t.Fatalf("resolve selector: %v", err)
} }
if len(overrides) != 1 { if got := referenceContractBindingLabels(overrides); !slices.Equal(got, tt.want) {
t.Fatalf("overrides = %#v, want one binding", overrides) t.Fatalf("binding targets = %#v, want %#v", got, tt.want)
}
for _, binding := range overrides {
if binding.Source != "reference.txt" || binding.BindingSource != contracts.ReferenceBindingSourceCLI {
t.Fatalf("binding = %#v, want CLI source", binding)
} }
got := overrides[0]
if got.Stage != tt.wantStage || got.LaneID != tt.wantLane || got.SlotName != tt.wantSlot || got.BindingSource != contracts.ReferenceBindingSourceCLI {
t.Fatalf("binding = %#v, want %s/%s/%s from CLI", got, tt.wantStage, tt.wantLane, tt.wantSlot)
} }
}) })
} }
} }
func TestReferenceSelectorsRejectAmbiguityWithSpecificSuggestions(t *testing.T) { func TestReferenceSelectorSpecificityAndFinalOccurrenceChooseConcreteBindings(t *testing.T) {
cfg := referenceContractConfig() cfg := referenceContractConfig()
catalog := referenceContractCatalog(t, true, true) catalog := referenceContractCatalog(t, true, true)
for _, tt := range []struct { requests := []cliReferenceRequest{
name string {Selector: mustParseReferenceSelector(t, "shared", "--reference"), Source: "pipeline-first.txt"},
selector string {Selector: mustParseReferenceSelector(t, "shared", "--reference"), Source: "pipeline-final.txt"},
want []string {Selector: mustParseReferenceSelector(t, "alpha.shared", "--reference"), Source: "lane.txt"},
}{ {Selector: mustParseReferenceSelector(t, "alpha.extract.shared", "--reference"), Source: "binding.txt"},
{name: "flat shared slot", selector: "shared", want: []string{"alpha.extract.shared", "beta.extract.shared"}}, }
{name: "lane shared slot", selector: "alpha.shared", want: []string{"alpha.extract.shared", "alpha.merge.shared", "alpha.normalize.shared"}}, overrides, unbinds, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog, requests, nil)
{name: "all mergers", selector: "merge.shared", want: []string{"alpha.merge.shared", "beta.merge.shared"}},
} {
t.Run(tt.name, func(t *testing.T) {
selector, err := parseReferenceSelector(tt.selector, "--reference")
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
_, _, err = resolveCLIReferenceRequests(cfg, "demo", nil, catalog, []cliReferenceRequest{{Selector: selector, Source: "reference.txt"}}, nil) if len(unbinds) != 0 {
if err == nil { t.Fatalf("unbinds = %#v, want none", unbinds)
t.Fatal("resolve selector succeeded, want ambiguity error")
} }
for _, fragment := range tt.want { want := map[string]string{
if !strings.Contains(err.Error(), fragment) { "alpha.extract.shared": "binding.txt",
t.Fatalf("error = %q, want suggestion %q", err, fragment) "alpha.merge.shared": "lane.txt",
"alpha.normalize.shared": "lane.txt",
"beta.extract.shared": "pipeline-final.txt",
"beta.merge.shared": "pipeline-final.txt",
"beta.normalize.shared": "pipeline-final.txt",
} }
for _, binding := range overrides {
label := referenceContractBindingLabel(binding)
if binding.Source != want[label] {
t.Fatalf("binding %s source = %q, want %q", label, binding.Source, want[label])
} }
}) delete(want, label)
}
if len(want) != 0 {
t.Fatalf("missing bindings: %#v", want)
} }
} }
func TestReferenceSelectorsRespectSelectedLanesBeforeMaterialization(t *testing.T) { func TestCompleteDNDSharedCLIReferencesExpandAcrossCompatibleTargets(t *testing.T) {
cfg := loadMaintainedExample(t, repositoryPath("examples", "dnd-complete.config.yml"))
catalog := catalogFromRegistries(productionTestComponents(t).registries)
sources := map[string]string{
"party": "/references/party.txt",
"players": "/references/players.txt",
"glossary": "/references/glossary.txt",
"spell_catalog": "/references/spells.json",
}
requests := make([]cliReferenceRequest, 0, len(sources))
for _, slot := range []string{"party", "players", "glossary", "spell_catalog"} {
requests = append(requests, cliReferenceRequest{
Selector: mustParseReferenceSelector(t, slot, "--reference"),
Source: sources[slot],
})
}
overrides, unbinds, err := resolveCLIReferenceRequests(cfg, "dnd-session", nil, catalog, requests, nil)
if err != nil {
t.Fatalf("expand complete D&D references: %v", err)
}
if len(unbinds) != 0 {
t.Fatalf("unbinds = %#v, want none", unbinds)
}
actual := make(map[string]pipeline.ReferenceBinding, len(overrides))
for _, binding := range overrides {
actual[referenceContractBindingLabel(binding)] = binding
}
targets, err := selectedReferenceTargets(cfg, "dnd-session", nil, catalog)
if err != nil {
t.Fatal(err)
}
matched := make(map[string]int, len(sources))
for _, target := range targets {
for slot, sourcePath := range sources {
if _, ok := target.slots[slot]; !ok {
continue
}
matched[slot]++
label := targetLabel(target) + "." + slot
binding, ok := actual[label]
if !ok || binding.Source != sourcePath || binding.BindingSource != contracts.ReferenceBindingSourceCLI {
t.Fatalf("binding %q = %#v, want CLI source %q", label, binding, sourcePath)
}
}
}
for slot := range sources {
if matched[slot] < 2 {
t.Fatalf("reference %q matched %d target(s), want a shared D&D reference", slot, matched[slot])
}
}
if _, err := cfg.Resolve(config.ResolveInput{PipelineID: "dnd-session", Catalog: catalog, ReferenceOverrides: overrides}); err != nil {
t.Fatalf("resolve complete D&D CLI references: %v", err)
}
}
func TestReferenceSelectorsRejectInvalidOrUnselectedScopesBeforeMaterialization(t *testing.T) {
cfg := referenceContractConfig() cfg := referenceContractConfig()
catalog := referenceContractCatalog(t, true, true) catalog := referenceContractCatalog(t, true, true)
for _, tt := range []struct { for _, tt := range []struct {
name string name string
selector string selector string
only []string
want string want string
}{ }{
{name: "unselected lane", selector: "beta.extract.beta-slot", want: `reference lane "beta" is not selected`}, {name: "pipeline slot", selector: "missing", want: `reference slot "missing" is not declared by any selected target`},
{name: "lane slot", selector: "alpha.missing", want: `reference slot "missing" is not declared by selected lane "alpha"`},
{name: "binding slot", selector: "alpha.extract.missing", want: `reference slot "missing" is not declared`},
{name: "former merge shorthand", selector: "merge.shared", want: `reference lane "merge" is not selected`},
{name: "unselected lane", selector: "beta.extract.beta-slot", only: []string{"alpha"}, want: `reference lane "beta" is not selected`},
{name: "unknown lane", selector: "missing.extract.beta-slot", want: `reference lane "missing" is not selected`}, {name: "unknown lane", selector: "missing.extract.beta-slot", want: `reference lane "missing" is not selected`},
} { } {
t.Run(tt.name, func(t *testing.T) { t.Run(tt.name, func(t *testing.T) {
selector, err := parseReferenceSelector(tt.selector, "--reference") selector := mustParseReferenceSelector(t, tt.selector, "--reference")
if err != nil { _, _, err := resolveCLIReferenceRequests(cfg, "demo", tt.only, catalog, []cliReferenceRequest{{Selector: selector, Source: filepath.Join(t.TempDir(), "missing.txt")}}, nil)
t.Fatal(err)
}
_, _, err = resolveCLIReferenceRequests(cfg, "demo", []string{"alpha"}, catalog, []cliReferenceRequest{{Selector: selector, Source: filepath.Join(t.TempDir(), "missing.txt")}}, nil)
if err == nil || !strings.Contains(err.Error(), tt.want) || strings.Contains(err.Error(), "missing.txt") { if err == nil || !strings.Contains(err.Error(), tt.want) || strings.Contains(err.Error(), "missing.txt") {
t.Fatalf("error = %v, want selection failure before file access", err) t.Fatalf("error = %v, want selection failure containing %q before file access", err, tt.want)
} }
}) })
} }
@@ -131,40 +193,50 @@ func TestReferenceSyntaxErrorsReturnTwo(t *testing.T) {
} }
} }
func TestReferenceOverridesUseFinalExactTargetBinding(t *testing.T) { func TestReferenceBindAndUnbindSpecificity(t *testing.T) {
cfg := referenceContractConfig() cfg := referenceContractConfig()
catalog := referenceContractCatalog(t, true, true) catalog := referenceContractCatalog(t, true, true)
alphaShared, err := parseReferenceSelector("alpha.extract.shared", "--reference") t.Run("specific unbind carves out broad binding", func(t *testing.T) {
overrides, unbinds, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog,
[]cliReferenceRequest{{Selector: mustParseReferenceSelector(t, "shared", "--reference"), Source: "shared.txt"}},
[]cliReferenceUnbindRequest{{Selector: mustParseReferenceSelector(t, "alpha.extract.shared", "--without-reference")}},
)
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
betaShared, err := parseReferenceSelector("beta.extract.shared", "--reference") if got := referenceContractBindingLabels(overrides); slices.Contains(got, "alpha.extract.shared") || len(got) != 5 {
t.Fatalf("overrides = %#v, want all shared targets except alpha extract", got)
}
if got := referenceContractUnbindLabels(unbinds); !slices.Equal(got, []string{"alpha.extract.shared"}) {
t.Fatalf("unbinds = %#v, want alpha extract", got)
}
})
t.Run("specific binding restores broad unbind", func(t *testing.T) {
overrides, unbinds, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog,
[]cliReferenceRequest{{Selector: mustParseReferenceSelector(t, "alpha.extract.shared", "--reference"), Source: "alpha.txt"}},
[]cliReferenceUnbindRequest{{Selector: mustParseReferenceSelector(t, "shared", "--without-reference")}},
)
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
overrides, unbinds, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog, []cliReferenceRequest{ if got := referenceContractBindingLabels(overrides); !slices.Equal(got, []string{"alpha.extract.shared"}) {
{Selector: alphaShared, Source: "alpha-first.txt"}, t.Fatalf("overrides = %#v, want alpha extract", got)
{Selector: alphaShared, Source: "alpha-final.txt"},
{Selector: betaShared, Source: "beta-only.txt"},
}, nil)
if err != nil {
t.Fatal(err)
} }
if len(unbinds) != 0 { if got := referenceContractUnbindLabels(unbinds); slices.Contains(got, "alpha.extract.shared") || len(got) != 5 {
t.Fatalf("unbinds = %#v, want none", unbinds) t.Fatalf("unbinds = %#v, want all shared targets except alpha extract", got)
} }
effective, err := cfg.Resolve(config.ResolveInput{PipelineID: "demo", Catalog: catalog, ReferenceOverrides: overrides}) })
if err != nil {
t.Fatalf("resolve pipeline: %v", err) t.Run("same specificity conflicts", func(t *testing.T) {
} _, _, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog,
alpha := referenceContractLane(t, effective.ResolvedPipeline, "alpha") []cliReferenceRequest{{Selector: mustParseReferenceSelector(t, "alpha.shared", "--reference"), Source: "alpha.txt"}},
beta := referenceContractLane(t, effective.ResolvedPipeline, "beta") []cliReferenceUnbindRequest{{Selector: mustParseReferenceSelector(t, "alpha.shared", "--without-reference")}},
if source := referenceContractBindingSource(alpha.ExtractReferences.Bindings, "shared"); source != "alpha-final.txt" { )
t.Fatalf("alpha shared source = %q, want final exact-target override", source) if err == nil || !strings.Contains(err.Error(), "same specificity") {
} t.Fatalf("error = %v, want same-specificity conflict", err)
if source := referenceContractBindingSource(beta.ExtractReferences.Bindings, "shared"); source != "beta-only.txt" {
t.Fatalf("beta shared source = %q, want target-specific override", source)
} }
})
} }
func TestReferenceUnbindsRemoveOptionalAndProtectRequiredSlots(t *testing.T) { func TestReferenceUnbindsRemoveOptionalAndProtectRequiredSlots(t *testing.T) {
@@ -257,6 +329,52 @@ func TestReferenceMaterializationSeparatesCLIAndConfigPathOrigins(t *testing.T)
} }
} }
func TestPipelineScopedCLIReferenceProtectsGeneratedHandoff(t *testing.T) {
cfg := referenceContractConfig()
profile := cfg.Pipelines["demo"]
alpha := profile.Artifacts["alpha"]
beta := profile.Artifacts["beta"]
alpha.Extract.References["shared"] = pipeline.GeneratedReference("produce", "beta")
profile.Artifacts = nil
profile.Steps = []pipeline.PipelineStepProfile{
{ID: "produce", Artifacts: map[string]pipeline.ArtifactLaneProfile{"beta": beta}},
{ID: "consume", Artifacts: map[string]pipeline.ArtifactLaneProfile{"alpha": alpha}},
}
cfg.Pipelines["demo"] = profile
catalog := referenceContractCatalog(t, true, true)
t.Run("binding conflicts before file access", func(t *testing.T) {
overrides, _, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog, []cliReferenceRequest{{
Selector: mustParseReferenceSelector(t, "shared", "--reference"),
Source: filepath.Join(t.TempDir(), "never-read.json"),
}}, nil)
if err != nil {
t.Fatalf("expand CLI reference: %v", err)
}
_, err = cfg.Resolve(config.ResolveInput{PipelineID: "demo", Catalog: catalog, ReferenceOverrides: overrides})
if err == nil || !strings.Contains(err.Error(), "conflicting generated and external bindings") || strings.Contains(err.Error(), "never-read.json") {
t.Fatalf("resolve error = %v, want generated/external conflict before file access", err)
}
})
t.Run("unbind leaves generated source intact", func(t *testing.T) {
_, unbinds, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog, nil, []cliReferenceUnbindRequest{{
Selector: mustParseReferenceSelector(t, "shared", "--without-reference"),
}})
if err != nil {
t.Fatalf("expand CLI unbind: %v", err)
}
effective, err := cfg.Resolve(config.ResolveInput{PipelineID: "demo", Catalog: catalog, ReferenceUnbinds: unbinds})
if err != nil {
t.Fatalf("resolve generated reference with CLI unbind: %v", err)
}
binding := referenceContractFindBinding(referenceContractLane(t, effective.ResolvedPipeline, "alpha").ExtractReferences.Bindings, "shared")
if binding == nil || binding.Artifact == nil || binding.Artifact.Step != "produce" || binding.Artifact.Lane != "beta" {
t.Fatalf("generated binding = %#v, want preserved produce/beta handoff", binding)
}
})
}
func TestReferenceTargetLookupUsesArtifactVariantsAndReportsMissingContext(t *testing.T) { func TestReferenceTargetLookupUsesArtifactVariantsAndReportsMissingContext(t *testing.T) {
cfg := referenceContractConfig() cfg := referenceContractConfig()
full := referenceContractCatalog(t, true, true) full := referenceContractCatalog(t, true, true)
@@ -368,7 +486,7 @@ func referenceContractCatalog(t *testing.T, includeBetaMerger, includeBetaNormal
register(registries.Chunkers.RegisterWithSpec(pipeline.ModuleSpec{Key: "reference/chunk", Stage: pipeline.StageChunk, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"source"}, Provides: []string{"chunks"}, ReferenceSlots: []contracts.ReferenceSlot{{Name: "chunk-slot"}, {Name: "required-chunk", Required: true}}}, func() (contracts.Chunker, error) { return stateTestChunker{}, nil })) register(registries.Chunkers.RegisterWithSpec(pipeline.ModuleSpec{Key: "reference/chunk", Stage: pipeline.StageChunk, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"source"}, Provides: []string{"chunks"}, ReferenceSlots: []contracts.ReferenceSlot{{Name: "chunk-slot"}, {Name: "required-chunk", Required: true}}}, func() (contracts.Chunker, error) { return stateTestChunker{}, nil }))
register(pipeline.RegisterArtifactCodec(registries.ArtifactCodecs, referenceContractCodecA{})) register(pipeline.RegisterArtifactCodec(registries.ArtifactCodecs, referenceContractCodecA{}))
register(pipeline.RegisterArtifactCodec(registries.ArtifactCodecs, referenceContractCodecB{})) register(pipeline.RegisterArtifactCodec(registries.ArtifactCodecs, referenceContractCodecB{}))
register(pipeline.RegisterExtractor(registries.Extractors, pipeline.ModuleSpec{Key: "reference/extract-alpha", Stage: pipeline.StageExtract, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: referenceContractKindAlpha, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "alpha-slot"}, {Name: "required-extract", Required: true}}}, func() (contracts.Extractor[stateTestArtifact], error) { return stateTestExtractor{}, nil })) register(pipeline.RegisterExtractor(registries.Extractors, pipeline.ModuleSpec{Key: "reference/extract-alpha", Stage: pipeline.StageExtract, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: referenceContractKindAlpha, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared", AcceptedArtifactKinds: []contracts.ArtifactKind{referenceContractKindBeta}, AcceptedMediaTypes: []string{"application/json"}}, {Name: "alpha-slot"}, {Name: "required-extract", Required: true}}}, func() (contracts.Extractor[stateTestArtifact], error) { return stateTestExtractor{}, nil }))
register(pipeline.RegisterExtractor(registries.Extractors, pipeline.ModuleSpec{Key: "reference/extract-beta", Stage: pipeline.StageExtract, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: referenceContractKindBeta, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "beta-slot"}, {Name: "required-extract", Required: true}}}, func() (contracts.Extractor[stateTestArtifact], error) { return stateTestExtractor{}, nil })) register(pipeline.RegisterExtractor(registries.Extractors, pipeline.ModuleSpec{Key: "reference/extract-beta", Stage: pipeline.StageExtract, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: referenceContractKindBeta, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "beta-slot"}, {Name: "required-extract", Required: true}}}, func() (contracts.Extractor[stateTestArtifact], error) { return stateTestExtractor{}, nil }))
register(pipeline.RegisterMerger(registries.Mergers, pipeline.ModuleSpec{Key: "reference/shared-merge", Stage: pipeline.StageMerge, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"artifact"}, Provides: []string{"merged"}, ArtifactKind: referenceContractKindAlpha, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "alpha-merge"}, {Name: "required-merge", Required: true}}}, func() (contracts.Merger[stateTestArtifact], error) { return stateTestMerger{}, nil })) register(pipeline.RegisterMerger(registries.Mergers, pipeline.ModuleSpec{Key: "reference/shared-merge", Stage: pipeline.StageMerge, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"artifact"}, Provides: []string{"merged"}, ArtifactKind: referenceContractKindAlpha, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "alpha-merge"}, {Name: "required-merge", Required: true}}}, func() (contracts.Merger[stateTestArtifact], error) { return stateTestMerger{}, nil }))
if includeBetaMerger { if includeBetaMerger {
@@ -444,6 +562,42 @@ func referenceContractBindingSource(bindings []pipeline.ReferenceBinding, slot s
return "" return ""
} }
func mustParseReferenceSelector(t *testing.T, value, flagName string) cliReferenceSelector {
t.Helper()
selector, err := parseReferenceSelector(value, flagName)
if err != nil {
t.Fatal(err)
}
return selector
}
func referenceContractBindingLabels(bindings []pipeline.ReferenceBinding) []string {
labels := make([]string, 0, len(bindings))
for _, binding := range bindings {
labels = append(labels, referenceContractBindingLabel(binding))
}
return labels
}
func referenceContractBindingLabel(binding pipeline.ReferenceBinding) string {
if binding.Stage == pipeline.StageChunk {
return "chunk." + binding.SlotName
}
return binding.LaneID + "." + string(binding.Stage) + "." + binding.SlotName
}
func referenceContractUnbindLabels(unbinds []pipeline.ReferenceUnbind) []string {
labels := make([]string, 0, len(unbinds))
for _, unbind := range unbinds {
labels = append(labels, referenceContractBindingLabel(pipeline.ReferenceBinding{
Stage: unbind.Stage,
LaneID: unbind.LaneID,
SlotName: unbind.SlotName,
}))
}
return labels
}
func referenceContractFindBinding(bindings []pipeline.ReferenceBinding, slot string) *pipeline.ReferenceBinding { func referenceContractFindBinding(bindings []pipeline.ReferenceBinding, slot string) *pipeline.ReferenceBinding {
for i := range bindings { for i := range bindings {
if bindings[i].SlotName == slot { if bindings[i].SlotName == slot {

View File

@@ -177,7 +177,7 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
fs.Var(&llmProfile, "llm-profile", "LLM profile override") fs.Var(&llmProfile, "llm-profile", "LLM profile override")
fs.Var(&reasoningEffort, "reasoning-effort", "reasoning effort override") fs.Var(&reasoningEffort, "reasoning-effort", "reasoning effort override")
fs.Var(&chunkCache, "chunk_cache", "chunk plan cache mode: auto, bypass, or refresh") fs.Var(&chunkCache, "chunk_cache", "chunk plan cache mode: auto, bypass, or refresh")
fs.Var(&referenceFlags, "reference", "reference binding, as slot=path, chunk.slot=path, merge.slot=path, lane.slot=path, lane.extract.slot=path, lane.merge.slot=path, or lane.normalize.slot=path") fs.Var(&referenceFlags, "reference", "reference binding, as slot=path, chunk.slot=path, lane.slot=path, lane.extract.slot=path, lane.merge.slot=path, or lane.normalize.slot=path")
fs.Var(&withoutReferenceFlags, "without-reference", "unbind a reference, using the same selector forms as --reference") fs.Var(&withoutReferenceFlags, "without-reference", "unbind a reference, using the same selector forms as --reference")
fs.Var(&recomputeStep, "recompute-step", "recompute one ordered pipeline step and dependent lanes") fs.Var(&recomputeStep, "recompute-step", "recompute one ordered pipeline step and dependent lanes")
if err := validateRunFlagValues(args); err != nil { if err := validateRunFlagValues(args); err != nil {
@@ -1287,11 +1287,21 @@ type cliReferenceUnbindRequest struct {
} }
type cliReferenceSelector struct { type cliReferenceSelector struct {
Scope cliReferenceSelectorScope
LaneID string LaneID string
Stage pipeline.ModuleStage Stage pipeline.ModuleStage
SlotName string SlotName string
} }
type cliReferenceSelectorScope uint8
const (
cliReferenceScopePipeline cliReferenceSelectorScope = iota
cliReferenceScopeLane
cliReferenceScopeChunk
cliReferenceScopeBinding
)
func parseReferenceFlags(values []string) ([]cliReferenceRequest, error) { func parseReferenceFlags(values []string) ([]cliReferenceRequest, error) {
if len(values) == 0 { if len(values) == 0 {
return nil, nil return nil, nil
@@ -1300,7 +1310,7 @@ func parseReferenceFlags(values []string) ([]cliReferenceRequest, error) {
for _, raw := range values { for _, raw := range values {
name, source, ok := strings.Cut(raw, "=") name, source, ok := strings.Cut(raw, "=")
if !ok { if !ok {
return nil, fmt.Errorf("--reference must use slot=path or lane.slot=path") return nil, fmt.Errorf("--reference must use slot=path, lane.slot=path, or lane.stage.slot=path")
} }
if strings.TrimSpace(source) == "" { if strings.TrimSpace(source) == "" {
return nil, fmt.Errorf("--reference path must not be empty; use --without-reference to unbind") return nil, fmt.Errorf("--reference path must not be empty; use --without-reference to unbind")
@@ -1350,17 +1360,14 @@ func parseReferenceSelector(raw string, flagName string) (cliReferenceSelector,
} }
switch len(parts) { switch len(parts) {
case 1: case 1:
return cliReferenceSelector{SlotName: strings.TrimSpace(parts[0])}, nil return cliReferenceSelector{Scope: cliReferenceScopePipeline, SlotName: strings.TrimSpace(parts[0])}, nil
case 2: case 2:
first := strings.TrimSpace(parts[0]) first := strings.TrimSpace(parts[0])
slotName := strings.TrimSpace(parts[1]) slotName := strings.TrimSpace(parts[1])
if first == string(pipeline.StageChunk) { if first == string(pipeline.StageChunk) {
return cliReferenceSelector{Stage: pipeline.StageChunk, SlotName: slotName}, nil return cliReferenceSelector{Scope: cliReferenceScopeChunk, Stage: pipeline.StageChunk, SlotName: slotName}, nil
} }
if first == string(pipeline.StageMerge) { return cliReferenceSelector{Scope: cliReferenceScopeLane, LaneID: first, SlotName: slotName}, nil
return cliReferenceSelector{Stage: pipeline.StageMerge, SlotName: slotName}, nil
}
return cliReferenceSelector{LaneID: first, SlotName: slotName}, nil
case 3: case 3:
laneID := strings.TrimSpace(parts[0]) laneID := strings.TrimSpace(parts[0])
stage := pipeline.ModuleStage(strings.TrimSpace(parts[1])) stage := pipeline.ModuleStage(strings.TrimSpace(parts[1]))
@@ -1368,9 +1375,9 @@ func parseReferenceSelector(raw string, flagName string) (cliReferenceSelector,
if stage != pipeline.StageExtract && stage != pipeline.StageMerge && stage != pipeline.StageNormalize { if stage != pipeline.StageExtract && stage != pipeline.StageMerge && stage != pipeline.StageNormalize {
return cliReferenceSelector{}, fmt.Errorf("%s lane-qualified selector must use lane.extract.slot, lane.merge.slot, or lane.normalize.slot", flagName) return cliReferenceSelector{}, fmt.Errorf("%s lane-qualified selector must use lane.extract.slot, lane.merge.slot, or lane.normalize.slot", flagName)
} }
return cliReferenceSelector{LaneID: laneID, Stage: stage, SlotName: slotName}, nil return cliReferenceSelector{Scope: cliReferenceScopeBinding, LaneID: laneID, Stage: stage, SlotName: slotName}, nil
default: default:
return cliReferenceSelector{}, fmt.Errorf("%s must use slot, chunk.slot, merge.slot, lane.slot, lane.extract.slot, lane.merge.slot, or lane.normalize.slot", flagName) return cliReferenceSelector{}, fmt.Errorf("%s must use slot, chunk.slot, lane.slot, lane.extract.slot, lane.merge.slot, or lane.normalize.slot", flagName)
} }
} }
@@ -1391,37 +1398,153 @@ func resolveCLIReferenceRequests(
return nil, nil, err return nil, nil, err
} }
overrides := make([]pipeline.ReferenceBinding, 0, len(referenceRequests)) // Broad CLI selectors are only presentation syntax. Collapse them into one
// highest-specificity action per concrete framework target before pipeline
// resolution so the generic reference contract stays stage-and-lane exact.
actions := make(map[cliReferenceTargetKey]resolvedCLIReferenceAction)
for _, request := range referenceRequests { for _, request := range referenceRequests {
target, err := resolveCLIReferenceTarget(targets, request.Selector) matches, err := resolveCLIReferenceTargets(targets, request.Selector)
if err != nil { if err != nil {
return nil, nil, err return nil, nil, err
} }
for _, target := range matches {
candidate := resolvedCLIReferenceAction{
kind: cliReferenceActionBind,
selector: request.Selector,
target: target,
slotName: request.Selector.SlotName,
source: request.Source,
specificity: request.Selector.specificity(),
}
if err := mergeCLIReferenceAction(actions, candidate); err != nil {
return nil, nil, err
}
}
}
for _, request := range unbindRequests {
matches, err := resolveCLIReferenceTargets(targets, request.Selector)
if err != nil {
return nil, nil, err
}
for _, target := range matches {
candidate := resolvedCLIReferenceAction{
kind: cliReferenceActionUnbind,
selector: request.Selector,
target: target,
slotName: request.Selector.SlotName,
specificity: request.Selector.specificity(),
}
if err := mergeCLIReferenceAction(actions, candidate); err != nil {
return nil, nil, err
}
}
}
resolved := make([]resolvedCLIReferenceAction, 0, len(actions))
for _, action := range actions {
resolved = append(resolved, action)
}
sort.Slice(resolved, func(i, j int) bool {
left, right := resolved[i], resolved[j]
if left.target.laneID != right.target.laneID {
return left.target.laneID < right.target.laneID
}
if left.target.stage != right.target.stage {
return referenceStageOrder(left.target.stage) < referenceStageOrder(right.target.stage)
}
return left.slotName < right.slotName
})
overrides := make([]pipeline.ReferenceBinding, 0, len(resolved))
unbinds := make([]pipeline.ReferenceUnbind, 0, len(resolved))
for _, action := range resolved {
switch action.kind {
case cliReferenceActionBind:
overrides = append(overrides, pipeline.ReferenceBinding{ overrides = append(overrides, pipeline.ReferenceBinding{
Stage: target.stage, Stage: action.target.stage,
LaneID: target.laneID, LaneID: action.target.laneID,
SlotName: request.Selector.SlotName, SlotName: action.slotName,
Source: request.Source, Source: action.source,
BindingSource: contracts.ReferenceBindingSourceCLI, BindingSource: contracts.ReferenceBindingSourceCLI,
}) })
} case cliReferenceActionUnbind:
unbinds := make([]pipeline.ReferenceUnbind, 0, len(unbindRequests))
for _, request := range unbindRequests {
target, err := resolveCLIReferenceTarget(targets, request.Selector)
if err != nil {
return nil, nil, err
}
unbinds = append(unbinds, pipeline.ReferenceUnbind{ unbinds = append(unbinds, pipeline.ReferenceUnbind{
Stage: target.stage, Stage: action.target.stage,
LaneID: target.laneID, LaneID: action.target.laneID,
SlotName: request.Selector.SlotName, SlotName: action.slotName,
}) })
} }
}
return overrides, unbinds, nil return overrides, unbinds, nil
} }
type cliReferenceActionKind uint8
const (
cliReferenceActionBind cliReferenceActionKind = iota
cliReferenceActionUnbind
)
type cliReferenceTargetKey struct {
stage pipeline.ModuleStage
laneID string
slotName string
}
type resolvedCLIReferenceAction struct {
kind cliReferenceActionKind
selector cliReferenceSelector
target selectedReferenceTarget
slotName string
source string
specificity int
}
func mergeCLIReferenceAction(actions map[cliReferenceTargetKey]resolvedCLIReferenceAction, candidate resolvedCLIReferenceAction) error {
key := cliReferenceTargetKey{stage: candidate.target.stage, laneID: candidate.target.laneID, slotName: candidate.slotName}
current, ok := actions[key]
if !ok || candidate.specificity > current.specificity {
actions[key] = candidate
return nil
}
if candidate.specificity < current.specificity {
return nil
}
if candidate.kind != current.kind {
return fmt.Errorf("reference target %q slot %q is both bound by %q and unbound by %q at the same specificity", targetLabel(candidate.target), candidate.slotName, current.selector.String(), candidate.selector.String())
}
actions[key] = candidate
return nil
}
func (selector cliReferenceSelector) specificity() int {
switch selector.Scope {
case cliReferenceScopePipeline:
return 0
case cliReferenceScopeLane:
return 1
case cliReferenceScopeChunk, cliReferenceScopeBinding:
return 2
default:
return -1
}
}
func (selector cliReferenceSelector) String() string {
switch selector.Scope {
case cliReferenceScopePipeline:
return selector.SlotName
case cliReferenceScopeLane:
return selector.LaneID + "." + selector.SlotName
case cliReferenceScopeChunk:
return "chunk." + selector.SlotName
case cliReferenceScopeBinding:
return selector.LaneID + "." + string(selector.Stage) + "." + selector.SlotName
default:
return selector.SlotName
}
}
type selectedReferenceTarget struct { type selectedReferenceTarget struct {
laneID string laneID string
stage pipeline.ModuleStage stage pipeline.ModuleStage
@@ -1638,68 +1761,39 @@ func referenceSlotSet(slots []contracts.ReferenceSlot) map[string]struct{} {
return slotSet return slotSet
} }
func resolveCLIReferenceTarget(targets []selectedReferenceTarget, selector cliReferenceSelector) (selectedReferenceTarget, error) { func resolveCLIReferenceTargets(targets []selectedReferenceTarget, selector cliReferenceSelector) ([]selectedReferenceTarget, error) {
slotName := strings.TrimSpace(selector.SlotName) slotName := strings.TrimSpace(selector.SlotName)
if slotName == "" { if slotName == "" {
return selectedReferenceTarget{}, fmt.Errorf("reference slot must not be empty") return nil, fmt.Errorf("reference slot must not be empty")
} }
if selector.Stage == pipeline.StageChunk { switch selector.Scope {
case cliReferenceScopePipeline:
matches := make([]selectedReferenceTarget, 0, len(targets))
for _, target := range targets {
if _, ok := target.slots[slotName]; ok {
matches = append(matches, target)
}
}
if len(matches) == 0 {
return nil, fmt.Errorf("reference slot %q is not declared by any selected target", slotName)
}
return matches, nil
case cliReferenceScopeChunk:
for _, target := range targets { for _, target := range targets {
if target.stage != pipeline.StageChunk { if target.stage != pipeline.StageChunk {
continue continue
} }
if _, ok := target.slots[slotName]; !ok { if _, ok := target.slots[slotName]; !ok {
return selectedReferenceTarget{}, fmt.Errorf("reference slot %q is not declared by chunk module %q", slotName, target.module) return nil, fmt.Errorf("reference slot %q is not declared by chunk module %q", slotName, target.module)
} }
return target, nil return []selectedReferenceTarget{target}, nil
} }
return selectedReferenceTarget{}, fmt.Errorf("reference chunk target is not selected") return nil, fmt.Errorf("reference chunk target is not selected")
} case cliReferenceScopeLane:
if selector.Stage == pipeline.StageExtract || selector.Stage == pipeline.StageMerge || selector.Stage == pipeline.StageNormalize {
if selector.LaneID == "" && selector.Stage == pipeline.StageMerge {
return resolveCLIReferenceStageTarget(targets, selector.Stage, slotName)
}
for _, target := range targets {
if target.laneID == selector.LaneID && target.stage == selector.Stage {
if _, ok := target.slots[slotName]; !ok {
return selectedReferenceTarget{}, fmt.Errorf("reference slot %q is not declared by selected %s target %q", slotName, selector.Stage, targetLabel(target))
}
return target, nil
}
}
return selectedReferenceTarget{}, fmt.Errorf("reference lane %q is not selected", selector.LaneID)
}
if strings.TrimSpace(selector.LaneID) != "" {
return resolveCLIReferenceLaneTarget(targets, strings.TrimSpace(selector.LaneID), slotName)
}
return resolveCLIReferenceFlatTarget(targets, slotName)
}
func resolveCLIReferenceStageTarget(targets []selectedReferenceTarget, stage pipeline.ModuleStage, slotName string) (selectedReferenceTarget, error) {
matches := make([]selectedReferenceTarget, 0, 2)
for _, target := range targets {
if target.stage != stage {
continue
}
if _, ok := target.slots[slotName]; ok {
matches = append(matches, target)
}
}
switch len(matches) {
case 0:
return selectedReferenceTarget{}, fmt.Errorf("reference slot %q is not declared by any selected %s target", slotName, stage)
case 1:
return matches[0], nil
default:
return selectedReferenceTarget{}, fmt.Errorf("reference slot %q is declared by multiple selected %s targets (%s); use a more specific selector such as %s", slotName, stage, targetList(matches), selectorSuggestions(matches, slotName))
}
}
func resolveCLIReferenceLaneTarget(targets []selectedReferenceTarget, laneID string, slotName string) (selectedReferenceTarget, error) {
laneSelected := false laneSelected := false
matches := make([]selectedReferenceTarget, 0, 2) matches := make([]selectedReferenceTarget, 0, 3)
for _, target := range targets { for _, target := range targets {
if target.laneID != laneID { if target.laneID != selector.LaneID {
continue continue
} }
laneSelected = true laneSelected = true
@@ -1708,44 +1802,36 @@ func resolveCLIReferenceLaneTarget(targets []selectedReferenceTarget, laneID str
} }
} }
if !laneSelected { if !laneSelected {
return selectedReferenceTarget{}, fmt.Errorf("reference lane %q is not selected", laneID) return nil, fmt.Errorf("reference lane %q is not selected", selector.LaneID)
} }
switch len(matches) { if len(matches) == 0 {
case 0: return nil, fmt.Errorf("reference slot %q is not declared by selected lane %q", slotName, selector.LaneID)
return selectedReferenceTarget{}, fmt.Errorf("reference slot %q is not declared by selected lane %q", slotName, laneID)
case 1:
return matches[0], nil
default:
return selectedReferenceTarget{}, fmt.Errorf("reference slot %q is declared by multiple selected targets in lane %q (%s); use a more specific selector such as %s", slotName, laneID, targetList(matches), selectorSuggestions(matches, slotName))
} }
} return matches, nil
case cliReferenceScopeBinding:
func resolveCLIReferenceFlatTarget(targets []selectedReferenceTarget, slotName string) (selectedReferenceTarget, error) { laneSelected := false
matches := make([]selectedReferenceTarget, 0, 2)
for _, target := range targets { for _, target := range targets {
if _, ok := target.slots[slotName]; ok { if target.laneID != selector.LaneID {
matches = append(matches, target) continue
} }
laneSelected = true
if target.stage != selector.Stage {
continue
} }
switch len(matches) { if _, ok := target.slots[slotName]; !ok {
case 0: return nil, fmt.Errorf("reference slot %q is not declared by selected %s target %q", slotName, selector.Stage, targetLabel(target))
return selectedReferenceTarget{}, fmt.Errorf("reference slot %q is not declared by any selected reference target", slotName) }
case 1: return []selectedReferenceTarget{target}, nil
return matches[0], nil }
if !laneSelected {
return nil, fmt.Errorf("reference lane %q is not selected", selector.LaneID)
}
return nil, fmt.Errorf("reference %s target is not selected for lane %q", selector.Stage, selector.LaneID)
default: default:
return selectedReferenceTarget{}, fmt.Errorf("reference slot %q is declared by multiple selected targets (%s); use a more specific selector such as %s", slotName, targetList(matches), selectorSuggestions(matches, slotName)) return nil, fmt.Errorf("reference selector has unknown scope")
} }
} }
func targetList(targets []selectedReferenceTarget) string {
labels := make([]string, 0, len(targets))
for _, target := range targets {
labels = append(labels, targetLabel(target))
}
sort.Strings(labels)
return strings.Join(labels, ", ")
}
func targetLabel(target selectedReferenceTarget) string { func targetLabel(target selectedReferenceTarget) string {
if target.stage == pipeline.StageChunk { if target.stage == pipeline.StageChunk {
return "chunk" return "chunk"
@@ -1753,17 +1839,19 @@ func targetLabel(target selectedReferenceTarget) string {
return target.laneID + "." + string(target.stage) return target.laneID + "." + string(target.stage)
} }
func selectorSuggestions(targets []selectedReferenceTarget, slotName string) string { func referenceStageOrder(stage pipeline.ModuleStage) int {
suggestions := make([]string, 0, len(targets)) switch stage {
for _, target := range targets { case pipeline.StageChunk:
if target.stage == pipeline.StageChunk { return 0
suggestions = append(suggestions, "chunk."+slotName) case pipeline.StageExtract:
continue return 1
case pipeline.StageMerge:
return 2
case pipeline.StageNormalize:
return 3
default:
return 4
} }
suggestions = append(suggestions, target.laneID+"."+string(target.stage)+"."+slotName)
}
sort.Strings(suggestions)
return strings.Join(suggestions, " or ")
} }
func sortedPipelineIDs(cfg config.Config) []string { func sortedPipelineIDs(cfg config.Config) []string {

View File

@@ -264,7 +264,7 @@ func TestChangedSemanticSpellCatalogFingerprintCannotResumeRecordedCheckpoint(t
if _, decision := changedLoader.Normalize("spells", spellnormalize.Key, normalizeDependencies); decision.Reused { if _, decision := changedLoader.Normalize("spells", spellnormalize.Key, normalizeDependencies); decision.Reused {
t.Fatalf("changed normalize fingerprint decision = %#v, want normalize checkpoint cold miss", decision) t.Fatalf("changed normalize fingerprint decision = %#v, want normalize checkpoint cold miss", decision)
} }
changedMapping := replaceCheckpointFingerprintValue(t, fingerprints, extractSpellMappingFingerprintName(), "dnd.spells.extract_mapping.v3") changedMapping := replaceCheckpointFingerprintValue(t, fingerprints, extractSpellMappingFingerprintName(), "dnd.spells.extract_mapping.changed")
assertOnlyCheckpointFingerprintChanged(t, fingerprints, changedMapping, extractSpellMappingFingerprintName()) assertOnlyCheckpointFingerprintChanged(t, fingerprints, changedMapping, extractSpellMappingFingerprintName())
_, mappingLoader, err := checkpointHandlersForRun(settings, Options{}, materialized, changedMapping, llmFingerprints, []byte("same input"), nil, nil, "", "", LLMRuntimeOverrides{}, true) _, mappingLoader, err := checkpointHandlersForRun(settings, Options{}, materialized, changedMapping, llmFingerprints, []byte("same input"), nil, nil, "", "", LLMRuntimeOverrides{}, true)
if err != nil { if err != nil {

View File

@@ -102,11 +102,12 @@ type TypedNormalizeResult[T any] struct {
ModelCandidate *ModelCandidate ModelCandidate *ModelCandidate
} }
// Normalize retry diagnostic limits bound module-provided values before the // Normalize retry limits bound module-provided control and diagnostic text
// framework persists them in debug artifacts. // before the framework consumes or records it.
const ( const (
MaxNormalizeRetryReasonCodeBytes = 128 MaxNormalizeRetryReasonCodeBytes = 128
MaxNormalizeRetryMessageBytes = 4096 MaxNormalizeRetryMessageBytes = 4096
MaxNormalizeRetryCorrectionGuidanceBytes = 4096
) )
// NormalizeRetry asks the framework to retry normalization while retaining a // NormalizeRetry asks the framework to retry normalization while retaining a
@@ -114,6 +115,7 @@ const (
type NormalizeRetry struct { type NormalizeRetry struct {
ReasonCode string ReasonCode string
Message string Message string
CorrectionGuidance string
FallbackDiagnostics []ProducerDiagnostic FallbackDiagnostics []ProducerDiagnostic
} }

View File

@@ -30,5 +30,14 @@ func validateNormalizeRetry(retry *contracts.NormalizeRetry) error {
if len(retry.Message) > contracts.MaxNormalizeRetryMessageBytes { if len(retry.Message) > contracts.MaxNormalizeRetryMessageBytes {
return errors.New("normalize retry directive message exceeds maximum length") return errors.New("normalize retry directive message exceeds maximum length")
} }
if !utf8.ValidString(retry.CorrectionGuidance) {
return errors.New("normalize retry directive correction guidance has invalid UTF-8")
}
if retry.CorrectionGuidance != "" && strings.TrimSpace(retry.CorrectionGuidance) == "" {
return errors.New("normalize retry directive correction guidance is blank")
}
if len(retry.CorrectionGuidance) > contracts.MaxNormalizeRetryCorrectionGuidanceBytes {
return errors.New("normalize retry directive correction guidance exceeds maximum length")
}
return nil return nil
} }

View File

@@ -101,6 +101,7 @@ func cloneNormalizeRetry(retry *contracts.NormalizeRetry) *contracts.NormalizeRe
return &contracts.NormalizeRetry{ return &contracts.NormalizeRetry{
ReasonCode: retry.ReasonCode, ReasonCode: retry.ReasonCode,
Message: retry.Message, Message: retry.Message,
CorrectionGuidance: retry.CorrectionGuidance,
FallbackDiagnostics: contracts.CloneProducerDiagnostics(retry.FallbackDiagnostics), FallbackDiagnostics: contracts.CloneProducerDiagnostics(retry.FallbackDiagnostics),
} }
} }

View File

@@ -21,6 +21,7 @@ func TestNormalizerRegistryErasureClonesRetryDirective(t *testing.T) {
retry := &contracts.NormalizeRetry{ retry := &contracts.NormalizeRetry{
ReasonCode: "retryable", ReasonCode: "retryable",
Message: "safe fallback available", Message: "safe fallback available",
CorrectionGuidance: "return a complete corrected proposal",
} }
registry := NewNormalizerRegistry() registry := NewNormalizerRegistry()
if err := RegisterNormalizer(registry, ModuleSpec{Key: "test/retry-normalize", Stage: StageNormalize, ExecutionClass: contracts.ExecutionClassDeterministic, ArtifactKind: "test/notes"}, func() (contracts.Normalizer[codecNotes], error) { if err := RegisterNormalizer(registry, ModuleSpec{Key: "test/retry-normalize", Stage: StageNormalize, ExecutionClass: contracts.ExecutionClassDeterministic, ArtifactKind: "test/notes"}, func() (contracts.Normalizer[codecNotes], error) {
@@ -41,7 +42,8 @@ func TestNormalizerRegistryErasureClonesRetryDirective(t *testing.T) {
t.Fatalf("normalize() error = %v", err) t.Fatalf("normalize() error = %v", err)
} }
retry.Message = "mutated" retry.Message = "mutated"
if result.Retry == nil || result.Retry.Message != "safe fallback available" { retry.CorrectionGuidance = "mutated guidance"
if result.Retry == nil || result.Retry.Message != "safe fallback available" || result.Retry.CorrectionGuidance != "return a complete corrected proposal" {
t.Fatalf("erased retry result = %#v, want independent retry data", result) t.Fatalf("erased retry result = %#v, want independent retry data", result)
} }
} }

View File

@@ -52,6 +52,7 @@ type producerAttemptRequest struct {
// the current value as a safe fallback if its shared budget is exhausted. // the current value as a safe fallback if its shared budget is exhausted.
// Artifact-specific adapters are responsible for validating and populating it. // Artifact-specific adapters are responsible for validating and populating it.
type producerRetryDirective struct { type producerRetryDirective struct {
CorrectionGuidance string
FallbackDiagnostics []contracts.ProducerDiagnostic FallbackDiagnostics []contracts.ProducerDiagnostic
} }
@@ -59,7 +60,10 @@ func (directive *producerRetryDirective) clone() *producerRetryDirective {
if directive == nil { if directive == nil {
return nil return nil
} }
return &producerRetryDirective{FallbackDiagnostics: contracts.CloneProducerDiagnostics(directive.FallbackDiagnostics)} return &producerRetryDirective{
CorrectionGuidance: directive.CorrectionGuidance,
FallbackDiagnostics: contracts.CloneProducerDiagnostics(directive.FallbackDiagnostics),
}
} }
// producerAttemptOutput is intentionally artifact-neutral. Value remains // producerAttemptOutput is intentionally artifact-neutral. Value remains
@@ -204,10 +208,21 @@ func runProducerAttempts(ctx context.Context, config producerAttemptConfig, prod
return failedProducerAttempt(provenance), err return failedProducerAttempt(provenance), err
} }
if output.Retry != nil && number < attemptLimit { if output.Retry != nil && number < attemptLimit {
correction, err = moduleRetryCorrection(output)
if err != nil {
provenance = append(provenance, producerAttemptProvenance{Number: number, Kind: kind, Outcome: producerAttemptFailed})
return failedProducerAttempt(provenance), err
}
provenance = append(provenance, producerAttemptProvenance{Number: number, Kind: kind, Outcome: producerAttemptRetried}) provenance = append(provenance, producerAttemptProvenance{Number: number, Kind: kind, Outcome: producerAttemptRetried})
kind, correction = producerAttemptModuleRetry, nil kind = producerAttemptModuleRetry
continue continue
} }
if output.Retry != nil && output.Retry.CorrectionGuidance != "" {
if _, err := moduleRetryCorrection(output); err != nil {
provenance = append(provenance, producerAttemptProvenance{Number: number, Kind: kind, Outcome: producerAttemptFailed})
return failedProducerAttempt(provenance), err
}
}
if output.Retry != nil { if output.Retry != nil {
output.Diagnostics = append(output.Diagnostics, contracts.CloneProducerDiagnostics(output.Retry.FallbackDiagnostics)...) output.Diagnostics = append(output.Diagnostics, contracts.CloneProducerDiagnostics(output.Retry.FallbackDiagnostics)...)
} }
@@ -269,6 +284,23 @@ func runProducerAttempts(ctx context.Context, config producerAttemptConfig, prod
return failedProducerAttempt(provenance), errors.New("producer attempt budget was not exhausted deterministically") return failedProducerAttempt(provenance), errors.New("producer attempt budget was not exhausted deterministically")
} }
func moduleRetryCorrection(output producerAttemptOutput) (*contracts.SemanticCorrection, error) {
if output.Retry == nil || output.Retry.CorrectionGuidance == "" {
return nil, nil
}
if output.Candidate == nil {
return nil, errors.New("module retry correction guidance requires a model candidate")
}
if output.Candidate.Protocol != contracts.CorrectionProtocolSingleResponseV1 {
return nil, fmt.Errorf("module retry correction guidance requires protocol %q", contracts.CorrectionProtocolSingleResponseV1)
}
correction, err := contracts.NewSemanticCorrection(output.Candidate.Response, output.Retry.CorrectionGuidance)
if err != nil {
return nil, fmt.Errorf("construct module retry semantic correction: %w", err)
}
return correction, nil
}
func validateProducerAttemptDiagnostics(output producerAttemptOutput) error { func validateProducerAttemptDiagnostics(output producerAttemptOutput) error {
if err := contracts.ValidateProducerDiagnostics(output.Diagnostics); err != nil { if err := contracts.ValidateProducerDiagnostics(output.Diagnostics); err != nil {
return fmt.Errorf("producer returned invalid diagnostics: %w", err) return fmt.Errorf("producer returned invalid diagnostics: %w", err)

View File

@@ -218,6 +218,9 @@ func TestRunProducerAttemptsUsesModuleRetryBudgetAndFallback(t *testing.T) {
calls := 0 calls := 0
terminal, err := runProducerAttempts(context.Background(), producerAttemptConfig{Retries: 1, Policy: DefaultValidationPolicy()}, func(_ context.Context, request producerAttemptRequest) (producerAttemptOutput, error) { terminal, err := runProducerAttempts(context.Background(), producerAttemptConfig{Retries: 1, Policy: DefaultValidationPolicy()}, func(_ context.Context, request producerAttemptRequest) (producerAttemptOutput, error) {
calls++ calls++
if request.Correction != nil {
t.Fatalf("feedback-free module retry correction = %#v, want nil", request.Correction)
}
if request.Number == 1 { if request.Number == 1 {
return producerAttemptOutput{Value: "fallback", Retry: &producerRetryDirective{}}, nil return producerAttemptOutput{Value: "fallback", Retry: &producerRetryDirective{}}, nil
} }
@@ -234,6 +237,58 @@ func TestRunProducerAttemptsUsesModuleRetryBudgetAndFallback(t *testing.T) {
} }
}) })
t.Run("feedback retry", func(t *testing.T) {
const (
defective = `{"duplicate_groups":[{"candidate_numbers":[1,99],"canonical_candidate_number":1}]}`
guidance = "Use only candidate numbers from the supplied candidate list. Return one complete corrected response."
)
var observed *contracts.SemanticCorrection
terminal, err := runProducerAttempts(context.Background(), producerAttemptConfig{Retries: 1, Policy: DefaultValidationPolicy()}, func(_ context.Context, request producerAttemptRequest) (producerAttemptOutput, error) {
if request.Number == 1 {
return producerAttemptOutput{Value: "safe fallback", Candidate: attemptCandidate(t, defective), Retry: &producerRetryDirective{CorrectionGuidance: guidance}}, nil
}
observed = request.Correction
return producerAttemptOutput{Value: "corrected"}, nil
}, approveAttempt)
if err != nil {
t.Fatalf("runProducerAttempts() error = %v", err)
}
if terminal.Action != producerTerminalAccepted || terminal.Value != "corrected" {
t.Fatalf("terminal = %#v, want corrected accepted value", terminal)
}
if observed == nil || string(observed.AssistantResponse) != defective || observed.UserGuidance != guidance {
t.Fatalf("module retry correction = %#v, want exact latest response and guidance", observed)
}
if got := attemptKinds(terminal.Provenance); !reflect.DeepEqual(got, []producerAttemptKind{producerAttemptInitial, producerAttemptModuleRetry}) {
t.Fatalf("attempt kinds = %v", got)
}
})
t.Run("feedback-free retry clears prior correction", func(t *testing.T) {
terminal, err := runProducerAttempts(context.Background(), producerAttemptConfig{Retries: 2, Policy: DefaultValidationPolicy()}, func(_ context.Context, request producerAttemptRequest) (producerAttemptOutput, error) {
switch request.Number {
case 1:
return producerAttemptOutput{Value: "first fallback", Candidate: attemptCandidate(t, "first defective response"), Retry: &producerRetryDirective{CorrectionGuidance: "Correct the first response."}}, nil
case 2:
if request.Correction == nil || string(request.Correction.AssistantResponse) != "first defective response" {
t.Fatalf("second attempt correction = %#v", request.Correction)
}
return producerAttemptOutput{Value: "second fallback", Retry: &producerRetryDirective{}}, nil
case 3:
if request.Correction != nil {
t.Fatalf("third attempt retained stale correction %#v", request.Correction)
}
return producerAttemptOutput{Value: "accepted"}, nil
default:
t.Fatalf("unexpected producer attempt %d", request.Number)
return producerAttemptOutput{}, nil
}
}, approveAttempt)
if err != nil || terminal.Action != producerTerminalAccepted || terminal.Value != "accepted" {
t.Fatalf("terminal = %#v, error = %v", terminal, err)
}
})
t.Run("fallback", func(t *testing.T) { t.Run("fallback", func(t *testing.T) {
fallbackDiagnostic := contracts.ProducerDiagnostic{Disposition: contracts.DiagnosticDispositionWarning, Category: contracts.DiagnosticCategoryFallback, ReasonCode: "fallback", OccurrenceCount: 1, Samples: []contracts.DiagnosticSample{{Scope: "fallback", Message: "fallback warning"}}} fallbackDiagnostic := contracts.ProducerDiagnostic{Disposition: contracts.DiagnosticDispositionWarning, Category: contracts.DiagnosticCategoryFallback, ReasonCode: "fallback", OccurrenceCount: 1, Samples: []contracts.DiagnosticSample{{Scope: "fallback", Message: "fallback warning"}}}
terminal, err := runProducerAttempts(context.Background(), producerAttemptConfig{Policy: DefaultValidationPolicy()}, func(context.Context, producerAttemptRequest) (producerAttemptOutput, error) { terminal, err := runProducerAttempts(context.Background(), producerAttemptConfig{Policy: DefaultValidationPolicy()}, func(context.Context, producerAttemptRequest) (producerAttemptOutput, error) {
@@ -248,6 +303,24 @@ func TestRunProducerAttemptsUsesModuleRetryBudgetAndFallback(t *testing.T) {
}) })
} }
func TestRunProducerAttemptsRejectsModuleCorrectionWithoutModelCandidate(t *testing.T) {
producerCalls := 0
validatorCalls := 0
terminal, err := runProducerAttempts(context.Background(), producerAttemptConfig{Retries: 1, Policy: DefaultValidationPolicy()}, func(context.Context, producerAttemptRequest) (producerAttemptOutput, error) {
producerCalls++
return producerAttemptOutput{Value: "safe", Retry: &producerRetryDirective{CorrectionGuidance: "Return a complete corrected response."}}, nil
}, func(context.Context, producerAttemptOutput) (validationReport, error) {
validatorCalls++
return validationReport{}, nil
})
if err == nil || !strings.Contains(err.Error(), "requires a model candidate") {
t.Fatalf("runProducerAttempts() error = %v, want model-candidate contract failure", err)
}
if terminal.Action != producerTerminalFailed || producerCalls != 1 || validatorCalls != 0 {
t.Fatalf("terminal = %#v, producer calls = %d, validator calls = %d", terminal, producerCalls, validatorCalls)
}
}
func TestRunProducerAttemptsRejectsInvalidDiagnosticsWithoutRetry(t *testing.T) { func TestRunProducerAttemptsRejectsInvalidDiagnosticsWithoutRetry(t *testing.T) {
tests := []struct { tests := []struct {
name string name string

View File

@@ -107,8 +107,12 @@ func (attemptDebugLLM) CompleteStructured(_ context.Context, request contracts.S
} }
func preparedAttemptDebugPipeline(t *testing.T) *PreparedPipeline { func preparedAttemptDebugPipeline(t *testing.T) *PreparedPipeline {
return preparedAttemptDebugPipelineWithChunks(t, 1)
}
func preparedAttemptDebugPipelineWithChunks(t *testing.T, chunkCount int) *PreparedPipeline {
t.Helper() t.Helper()
prepared := preparedConcurrentPipeline(t, 1) prepared := preparedConcurrentPipeline(t, chunkCount)
prepared.Steps[0].lanes = prepared.Steps[0].lanes[:1] prepared.Steps[0].lanes = prepared.Steps[0].lanes[:1]
prepared.resolved.Steps[0].ArtifactLanes = prepared.resolved.Steps[0].ArtifactLanes[:1] prepared.resolved.Steps[0].ArtifactLanes = prepared.resolved.Steps[0].ArtifactLanes[:1]
prepared.Steps[0].ArtifactLanes = prepared.Steps[0].ArtifactLanes[:1] prepared.Steps[0].ArtifactLanes = prepared.Steps[0].ArtifactLanes[:1]

View File

@@ -153,6 +153,60 @@ func TestRunnerHandlesRetryableNormalizeFallbacks(t *testing.T) {
} }
} }
func TestRunnerForwardsModuleRequestedNormalizeCorrection(t *testing.T) {
const (
defective = `{"duplicate_groups":[{"candidate_numbers":[1,99],"canonical_candidate_number":1}]}`
guidance = "Duplicate group 1 must use only supplied candidate numbers. Return one complete corrected response."
)
prepared := preparedAttemptDebugPipeline(t)
lane := &prepared.Steps[0].lanes[0]
lane.resolved.Normalize.Retries = 1
var observed *contracts.SemanticCorrection
calls := 0
lane.typed.normalize = func(_ context.Context, _ any, request contracts.TypedNormalizeRequest[any]) (erasedTypedResult, error) {
calls++
if request.Correction != nil {
clone, err := contracts.CloneSemanticCorrection(request.Correction)
if err != nil {
return erasedTypedResult{}, err
}
observed = clone
}
if calls == 1 {
return erasedTypedResult{
Value: codecNotes{Items: []string{"safe fallback"}},
ModelCandidate: attemptCandidate(t, defective),
Retry: &contracts.NormalizeRetry{
ReasonCode: "semantic_proposal_invalid",
Message: "operator-facing proposal diagnostic",
CorrectionGuidance: guidance,
},
}, nil
}
return erasedTypedResult{Value: codecNotes{Items: []string{"corrected"}}}, nil
}
debug := newCapturedDebugRecorder()
output, err := New().Run(context.Background(), RunInput{Prepared: prepared, RawInput: []byte("input"), Debug: debug})
if err != nil {
t.Fatalf("Run() error = %v", err)
}
if calls != 2 || observed == nil || string(observed.AssistantResponse) != defective || observed.UserGuidance != guidance {
t.Fatalf("normalize calls = %d correction = %#v", calls, observed)
}
if len(output.NormalizeOutputs) != 1 || len(output.Rejected) != 0 {
t.Fatalf("run output = %#v, want corrected accepted output", output)
}
var retryDebug strings.Builder
for _, name := range debug.names() {
if strings.HasPrefix(name, "normalize/notes/attempt-") && strings.HasSuffix(name, ".json") {
retryDebug.Write(debug.json[name])
}
}
if !strings.Contains(retryDebug.String(), `"correction_available":true`) || strings.Contains(retryDebug.String(), guidance) || strings.Contains(retryDebug.String(), defective) {
t.Fatalf("retry debug = %s, want safe correction metadata without content", retryDebug.String())
}
}
func TestRunnerValidatesNormalizeRetryDiagnostics(t *testing.T) { func TestRunnerValidatesNormalizeRetryDiagnostics(t *testing.T) {
const ( const (
reasonSentinel = "reason-diagnostic-sentinel" reasonSentinel = "reason-diagnostic-sentinel"
@@ -160,9 +214,11 @@ func TestRunnerValidatesNormalizeRetryDiagnostics(t *testing.T) {
) )
reasonOverLimit := strings.Repeat("r", contracts.MaxNormalizeRetryReasonCodeBytes-len(reasonSentinel)) + reasonSentinel + "x" reasonOverLimit := strings.Repeat("r", contracts.MaxNormalizeRetryReasonCodeBytes-len(reasonSentinel)) + reasonSentinel + "x"
messageOverLimit := strings.Repeat("m", contracts.MaxNormalizeRetryMessageBytes-len(messageSentinel)) + messageSentinel + "x" messageOverLimit := strings.Repeat("m", contracts.MaxNormalizeRetryMessageBytes-len(messageSentinel)) + messageSentinel + "x"
guidanceOverLimit := strings.Repeat("g", contracts.MaxNormalizeRetryCorrectionGuidanceBytes-len(messageSentinel)) + messageSentinel + "x"
tests := []struct { tests := []struct {
name string name string
retry contracts.NormalizeRetry retry contracts.NormalizeRetry
candidate *contracts.ModelCandidate
wantError string wantError string
hiddenValues []string hiddenValues []string
}{ }{
@@ -173,6 +229,15 @@ func TestRunnerValidatesNormalizeRetryDiagnostics(t *testing.T) {
Message: strings.Repeat("m", contracts.MaxNormalizeRetryMessageBytes), Message: strings.Repeat("m", contracts.MaxNormalizeRetryMessageBytes),
}, },
}, },
{
name: "accepts correction guidance byte limit with candidate",
retry: contracts.NormalizeRetry{
ReasonCode: reasonSentinel,
Message: messageSentinel,
CorrectionGuidance: strings.Repeat("g", contracts.MaxNormalizeRetryCorrectionGuidanceBytes),
},
candidate: attemptCandidate(t, `{"duplicate_groups":[]}`),
},
{ {
name: "rejects oversized reason code", name: "rejects oversized reason code",
retry: contracts.NormalizeRetry{ retry: contracts.NormalizeRetry{
@@ -209,6 +274,46 @@ func TestRunnerValidatesNormalizeRetryDiagnostics(t *testing.T) {
wantError: "message has invalid UTF-8", wantError: "message has invalid UTF-8",
hiddenValues: []string{reasonSentinel, messageSentinel}, hiddenValues: []string{reasonSentinel, messageSentinel},
}, },
{
name: "rejects oversized correction guidance",
retry: contracts.NormalizeRetry{
ReasonCode: reasonSentinel,
Message: messageSentinel,
CorrectionGuidance: guidanceOverLimit,
},
wantError: "correction guidance exceeds maximum length",
hiddenValues: []string{reasonSentinel, messageSentinel},
},
{
name: "rejects invalid correction guidance UTF-8",
retry: contracts.NormalizeRetry{
ReasonCode: reasonSentinel,
Message: messageSentinel,
CorrectionGuidance: messageSentinel + string([]byte{0xff}),
},
wantError: "correction guidance has invalid UTF-8",
hiddenValues: []string{reasonSentinel, messageSentinel},
},
{
name: "rejects blank correction guidance",
retry: contracts.NormalizeRetry{
ReasonCode: reasonSentinel,
Message: messageSentinel,
CorrectionGuidance: " \t\n ",
},
wantError: "correction guidance is blank",
hiddenValues: []string{reasonSentinel, messageSentinel},
},
{
name: "rejects correction guidance without candidate",
retry: contracts.NormalizeRetry{
ReasonCode: reasonSentinel,
Message: messageSentinel,
CorrectionGuidance: "Return one complete corrected response.",
},
wantError: "requires a model candidate",
hiddenValues: []string{reasonSentinel, messageSentinel},
},
{ {
name: "rejects blank reason code", name: "rejects blank reason code",
retry: contracts.NormalizeRetry{ retry: contracts.NormalizeRetry{
@@ -233,7 +338,7 @@ func TestRunnerValidatesNormalizeRetryDiagnostics(t *testing.T) {
t.Run(tc.name, func(t *testing.T) { t.Run(tc.name, func(t *testing.T) {
prepared := preparedAttemptDebugPipeline(t) prepared := preparedAttemptDebugPipeline(t)
prepared.Steps[0].lanes[0].typed.normalize = func(context.Context, any, contracts.TypedNormalizeRequest[any]) (erasedTypedResult, error) { prepared.Steps[0].lanes[0].typed.normalize = func(context.Context, any, contracts.TypedNormalizeRequest[any]) (erasedTypedResult, error) {
return erasedTypedResult{Value: codecNotes{Items: []string{"safe"}}, Retry: &tc.retry}, nil return erasedTypedResult{Value: codecNotes{Items: []string{"safe"}}, Retry: &tc.retry, ModelCandidate: tc.candidate}, nil
} }
debug := newCapturedDebugRecorder() debug := newCapturedDebugRecorder()
output, err := New().Run(context.Background(), RunInput{Prepared: prepared, RawInput: []byte("input"), Debug: debug}) output, err := New().Run(context.Background(), RunInput{Prepared: prepared, RawInput: []byte("input"), Debug: debug})

View File

@@ -3,6 +3,7 @@ package pipeline
import ( import (
"context" "context"
"errors" "errors"
"fmt"
"reflect" "reflect"
"sort" "sort"
"strconv" "strconv"
@@ -427,7 +428,7 @@ func TestRunnerDoesNotRetryAfterTerminalAttemptWriteFailure(t *testing.T) {
} }
func TestRunnerKeepsExtractModuleAndValidatorLLMCallsIsolated(t *testing.T) { func TestRunnerKeepsExtractModuleAndValidatorLLMCallsIsolated(t *testing.T) {
prepared := preparedAttemptDebugPipeline(t) prepared := preparedAttemptDebugPipelineWithChunks(t, 2)
debug := newCapturedDebugRecorder() debug := newCapturedDebugRecorder()
client := WithDebugLLMRecording(attemptDebugLLM{}, debug) client := WithDebugLLMRecording(attemptDebugLLM{}, debug)
installExtractOperation(prepared, 0, func(ctx context.Context, request contracts.TypedExtractionRequest) (erasedTypedResult, error) { installExtractOperation(prepared, 0, func(ctx context.Context, request contracts.TypedExtractionRequest) (erasedTypedResult, error) {
@@ -449,15 +450,23 @@ func TestRunnerKeepsExtractModuleAndValidatorLLMCallsIsolated(t *testing.T) {
if _, err := New().Run(context.Background(), RunInput{Prepared: prepared, RawInput: []byte("input"), Debug: debug}); err != nil { if _, err := New().Run(context.Background(), RunInput{Prepared: prepared, RawInput: []byte("input"), Debug: debug}); err != nil {
t.Fatalf("Run() error = %v, want nil", err) t.Fatalf("Run() error = %v, want nil", err)
} }
module := debug.envelope(t, "extract/notes/chunk-000001/attempt-01.json") validatorResponses := make(map[string]struct{})
validator := debug.envelope(t, "validate/extract/notes/typed%2Fextract-notes/01-llm-check-attempt-01.json") for chunkNumber := 1; chunkNumber <= 2; chunkNumber++ {
if len(module.LLMCalls) != 1 || !strings.Contains(module.LLMCalls[0].ResponsePath, "extract/notes/chunk-000001/attempt-01/") { chunkPath := fmt.Sprintf("chunk-%06d", chunkNumber)
t.Fatalf("module LLM calls = %#v, want extract module call only", module.LLMCalls) module := debug.envelope(t, "extract/notes/"+chunkPath+"/attempt-01.json")
validator := debug.envelope(t, "validate/extract/notes/typed%2Fextract-notes/"+chunkPath+"/01-llm-check-attempt-01.json")
if len(module.LLMCalls) != 1 || !strings.Contains(module.LLMCalls[0].ResponsePath, "extract/notes/"+chunkPath+"/attempt-01/") {
t.Fatalf("module LLM calls for %s = %#v, want extract module call only", chunkPath, module.LLMCalls)
} }
if len(validator.LLMCalls) != 1 || !strings.Contains(validator.LLMCalls[0].ResponsePath, "validate/extract/notes/") { if len(validator.LLMCalls) != 1 || !strings.Contains(validator.LLMCalls[0].ResponsePath, "validate/extract/notes/typed%2Fextract-notes/"+chunkPath+"/") {
t.Fatalf("validator LLM calls = %#v, want validator call only", validator.LLMCalls) t.Fatalf("validator LLM calls for %s = %#v, want chunk-scoped validator call only", chunkPath, validator.LLMCalls)
} }
if module.LLMCalls[0].CallID == validator.LLMCalls[0].CallID { if module.LLMCalls[0].CallID == validator.LLMCalls[0].CallID {
t.Fatalf("module and validator attempts share LLM call %#v", module.LLMCalls) t.Fatalf("module and validator attempts for %s share LLM call %#v", chunkPath, module.LLMCalls)
}
validatorResponses[validator.LLMCalls[0].ResponsePath] = struct{}{}
}
if len(validatorResponses) != 2 {
t.Fatalf("validator response paths = %#v, want one distinct path per chunk", validatorResponses)
} }
} }

View File

@@ -451,8 +451,8 @@ func (r *Runner) runNormalizeStage(ctx context.Context, input RunInput, checkpoi
return producerAttemptOutput{}, terminal.record(map[string]any{"output": debugCheckpointArtifact(serializedCandidate)}, fmt.Errorf("normalize lane %q returned invalid retry directive: %w", lane.ID, err)) return producerAttemptOutput{}, terminal.record(map[string]any{"output": debugCheckpointArtifact(serializedCandidate)}, fmt.Errorf("normalize lane %q returned invalid retry directive: %w", lane.ID, err))
} }
anotherAttempt := request.Number <= lane.Normalize.Retries anotherAttempt := request.Number <= lane.Normalize.Retries
attemptValue.retry = map[string]any{"reason_code": result.Retry.ReasonCode, "message": result.Retry.Message, "another_attempt": anotherAttempt, "fallback_accepted": !anotherAttempt} attemptValue.retry = map[string]any{"reason_code": result.Retry.ReasonCode, "message": result.Retry.Message, "correction_available": result.Retry.CorrectionGuidance != "", "another_attempt": anotherAttempt, "fallback_accepted": !anotherAttempt}
directive = &producerRetryDirective{FallbackDiagnostics: contracts.CloneProducerDiagnostics(result.Retry.FallbackDiagnostics)} directive = &producerRetryDirective{CorrectionGuidance: result.Retry.CorrectionGuidance, FallbackDiagnostics: contracts.CloneProducerDiagnostics(result.Retry.FallbackDiagnostics)}
if anotherAttempt { if anotherAttempt {
payload := map[string]any{"output": debugCheckpointArtifact(serializedCandidate), "retry": attemptValue.retry} payload := map[string]any{"output": debugCheckpointArtifact(serializedCandidate), "retry": attemptValue.retry}
if debugErr := terminal.record(payload, nil); debugErr != nil { if debugErr := terminal.record(payload, nil); debugErr != nil {
@@ -596,7 +596,12 @@ func (r *Runner) validateTypedReport(ctx context.Context, codec artifactCodecEnt
var result contracts.ValidationResult var result contracts.ValidationResult
var err error var err error
started := time.Now().UTC() started := time.Now().UTC()
attemptPath := validatorAttemptPath(path.Join("validate", fileio.EncodePathComponent(string(target.stage)), fileio.EncodePathComponent(target.laneID), fileio.EncodePathComponent(target.moduleKey), fmt.Sprintf("%02d-%s-attempt-%02d", item.position, fileio.EncodePathComponent(binding.Module), attempt)), validatorAttempt) pathParts := []string{"validate", fileio.EncodePathComponent(string(target.stage)), fileio.EncodePathComponent(target.laneID), fileio.EncodePathComponent(target.moduleKey)}
if target.chunk != nil {
pathParts = append(pathParts, fmt.Sprintf("chunk-%06d", target.chunk.Index+1))
}
pathParts = append(pathParts, fmt.Sprintf("%02d-%s-attempt-%02d", item.position, fileio.EncodePathComponent(binding.Module), attempt))
attemptPath := validatorAttemptPath(path.Join(pathParts...), validatorAttempt)
validatorCtx, llmScope := withIsolatedDebugLLMScope(validatorCtx, attemptPath) validatorCtx, llmScope := withIsolatedDebugLLMScope(validatorCtx, attemptPath)
requestTarget := target requestTarget := target
requestTarget.sourceInput = target.sourceInput.Clone() requestTarget.sourceInput = target.sourceInput.Clone()

View File

@@ -31,7 +31,7 @@ func TestRegisterAssetsPreparesGenericPromptOffline(t *testing.T) {
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{ prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{
PromptID: PromptID, PromptVersion: PromptVersion, ProfileID: "semantic-reconciliation-test", PromptID: PromptID, PromptVersion: PromptVersion, ProfileID: "semantic-reconciliation-test",
Inputs: map[string]promptkit.ArtifactRef{ Inputs: map[string]promptkit.ArtifactRef{
"candidates": promptkit.Inline(`{"candidates":[{"candidate_id":1,"label":"Mira"},{"candidate_id":2,"label":"Captain Mira"}]}`), "candidates": promptkit.Inline(`{"candidate_number_range":{"first":1,"last":2},"candidates":[{"candidate_number":1,"label":"Mira"},{"candidate_number":2,"label":"Captain Mira"}]}`),
"transcript": promptkit.Inline(`{"windows":[{"units":[{"unit_id":7,"text":"Mira arrived."}]}]}`), "transcript": promptkit.Inline(`{"windows":[{"units":[{"unit_id":7,"text":"Mira arrived."}]}]}`),
}, },
}) })
@@ -60,15 +60,15 @@ func TestRegisterAssetsPreparesGenericPromptOffline(t *testing.T) {
} }
} }
protocol := prepared.Messages[1].Content protocol := prepared.Messages[1].Content
for _, requirement := range []string{"positive integer", "Return IDs only", "do not copy candidate names", "source ranges"} { for _, requirement := range []string{"candidate_number_range", "Return candidate numbers only", "do not copy candidate names", "start_unit_id", "not candidate numbers"} {
if !strings.Contains(protocol, requirement) { if !strings.Contains(protocol, requirement) {
t.Fatalf("protocol message = %q, want requirement %q", protocol, requirement) t.Fatalf("protocol message = %q, want requirement %q", protocol, requirement)
} }
} }
if !strings.Contains(prepared.Messages[3].Content, `"candidate_id":1`) || strings.Contains(prepared.Messages[3].Content, `"windows"`) { if !strings.Contains(prepared.Messages[3].Content, `"candidate_number":1`) || strings.Contains(prepared.Messages[3].Content, `"windows"`) {
t.Fatalf("candidate message = %q, want only integer candidate material", prepared.Messages[3].Content) t.Fatalf("candidate message = %q, want only integer candidate material", prepared.Messages[3].Content)
} }
if !strings.Contains(prepared.Messages[4].Content, `"windows"`) || strings.Contains(prepared.Messages[4].Content, `"candidate_id"`) { if !strings.Contains(prepared.Messages[4].Content, `"windows"`) || strings.Contains(prepared.Messages[4].Content, `"candidate_number"`) {
t.Fatalf("transcript message = %q, want only transcript windows", prepared.Messages[4].Content) t.Fatalf("transcript message = %q, want only transcript windows", prepared.Messages[4].Content)
} }
} }

View File

@@ -69,6 +69,7 @@ type Result struct {
plan Plan plan Plan
issues []Issue issues []Issue
discardedGroupCount int discardedGroupCount int
candidateNumberRange CandidateNumberRange
candidateMappings []CandidateMapping candidateMappings []CandidateMapping
modelCandidate *contracts.ModelCandidate modelCandidate *contracts.ModelCandidate
} }
@@ -85,11 +86,27 @@ func (result Result) Issues() []Issue { return append([]Issue(nil), result.issue
// DiscardedGroupCount returns the number of excluded proposal groups. // DiscardedGroupCount returns the number of excluded proposal groups.
func (result Result) DiscardedGroupCount() int { return result.discardedGroupCount } func (result Result) DiscardedGroupCount() int { return result.discardedGroupCount }
// CandidateMappings returns the request-local handle mapping used for this call. // CandidateNumberRange returns the exact request-local range used for this call.
func (result Result) CandidateNumberRange() CandidateNumberRange {
return result.candidateNumberRange
}
// CandidateMappings returns the request-local number mapping used for this call.
func (result Result) CandidateMappings() []CandidateMapping { func (result Result) CandidateMappings() []CandidateMapping {
return append([]CandidateMapping(nil), result.candidateMappings...) return append([]CandidateMapping(nil), result.candidateMappings...)
} }
// CorrectionDetails returns bounded model-facing descriptions of the result's
// proposal issues using the exact candidate-number range for this call.
func (result Result) CorrectionDetails() ([]string, error) {
return CorrectionDetails(result.Issues(), result.candidateNumberRange)
}
// CorrectionGuidance returns one bounded replacement request for this result.
func (result Result) CorrectionGuidance(additionalDetails ...string) (string, error) {
return CorrectionGuidance(result.Issues(), result.candidateNumberRange, additionalDetails...)
}
// ModelCandidate returns an owned copy of the proposal response when a model // ModelCandidate returns an owned copy of the proposal response when a model
// completion produced this result. // completion produced this result.
func (result Result) ModelCandidate() *contracts.ModelCandidate { func (result Result) ModelCandidate() *contracts.ModelCandidate {
@@ -158,7 +175,10 @@ func (engine *Engine) Reconcile(ctx context.Context, request Request) (Result, e
if err != nil { if err != nil {
return Result{}, fmt.Errorf("semantic reconciliation %q: prepare materials: %w", request.StageName, err) return Result{}, fmt.Errorf("semantic reconciliation %q: prepare materials: %w", request.StageName, err)
} }
result := Result{candidateMappings: preparation.CandidateMappings()} result := Result{
candidateNumberRange: preparation.CandidateNumberRange(),
candidateMappings: preparation.CandidateMappings(),
}
switch preparation.Disposition() { switch preparation.Disposition() {
case InsufficientCandidates: case InsufficientCandidates:
result.disposition = SkippedInsufficientCandidates result.disposition = SkippedInsufficientCandidates

View File

@@ -50,7 +50,7 @@ func TestNewEngineValidatesConstruction(t *testing.T) {
func TestEnginePropagatesRequestAndAssessesResponse(t *testing.T) { func TestEnginePropagatesRequestAndAssessesResponse(t *testing.T) {
client := &recordingReconciliationClient{responses: []ProposalResponse{{DuplicateGroups: []DuplicateGroup{{ client := &recordingReconciliationClient{responses: []ProposalResponse{{DuplicateGroups: []DuplicateGroup{{
CandidateIDs: []int{1, 2}, CanonicalCandidateID: 2, CandidateNumbers: []int{1, 2}, CanonicalCandidateNumber: 2,
}}}}} }}}}}
engine := newTestEngine(t, client, DefaultLimits()) engine := newTestEngine(t, client, DefaultLimits())
request := readyEngineRequest() request := readyEngineRequest()
@@ -69,6 +69,9 @@ func TestEnginePropagatesRequestAndAssessesResponse(t *testing.T) {
if result.Disposition() != Complete || result.DiscardedGroupCount() != 0 || len(result.Issues()) != 0 { if result.Disposition() != Complete || result.DiscardedGroupCount() != 0 || len(result.Issues()) != 0 {
t.Fatalf("result = disposition %v discarded %d issues %#v", result.Disposition(), result.DiscardedGroupCount(), result.Issues()) t.Fatalf("result = disposition %v discarded %d issues %#v", result.Disposition(), result.DiscardedGroupCount(), result.Issues())
} }
if got, want := result.CandidateNumberRange(), (CandidateNumberRange{First: 1, Last: 2}); got != want {
t.Fatalf("candidate number range = %#v, want %#v", got, want)
}
groups := result.Plan().Groups() groups := result.Plan().Groups()
if len(groups) != 1 || !reflect.DeepEqual(groups[0].MemberPositions(), []int{0, 1}) || groups[0].CanonicalPosition() != 1 { if len(groups) != 1 || !reflect.DeepEqual(groups[0].MemberPositions(), []int{0, 1}) || groups[0].CanonicalPosition() != 1 {
t.Fatalf("safe plan = %#v", groups) t.Fatalf("safe plan = %#v", groups)
@@ -105,7 +108,7 @@ func TestEngineClassifiesSemanticAndTransportOutcomes(t *testing.T) {
wantError error wantError error
}{ }{
{name: "empty groups complete", response: ProposalResponse{DuplicateGroups: []DuplicateGroup{}}, want: Complete, wantCandidate: true}, {name: "empty groups complete", response: ProposalResponse{DuplicateGroups: []DuplicateGroup{}}, want: Complete, wantCandidate: true},
{name: "discarded proposal retryable", response: ProposalResponse{DuplicateGroups: []DuplicateGroup{{CandidateIDs: []int{1, 99}, CanonicalCandidateID: 1}}}, want: RetryableDiscardedProposalGroups, wantDiscard: 1, wantIssues: true, wantCandidate: true}, {name: "discarded proposal retryable", response: ProposalResponse{DuplicateGroups: []DuplicateGroup{{CandidateNumbers: []int{1, 99}, CanonicalCandidateNumber: 1}}}, want: RetryableDiscardedProposalGroups, wantDiscard: 1, wantIssues: true, wantCandidate: true},
{name: "invalid structured output retryable", completion: fmt.Errorf("decode response: %w", contracts.ErrInvalidStructuredOutput), want: RetryableInvalidStructuredOutput}, {name: "invalid structured output retryable", completion: fmt.Errorf("decode response: %w", contracts.ErrInvalidStructuredOutput), want: RetryableInvalidStructuredOutput},
{name: "transport failure", completion: transportErr, wantError: transportErr}, {name: "transport failure", completion: transportErr, wantError: transportErr},
} }
@@ -213,8 +216,8 @@ func TestEngineRejectsInvalidInvocationAndHonorsCancellation(t *testing.T) {
func TestEngineCallsAreIndependentAndResultsAreOwned(t *testing.T) { func TestEngineCallsAreIndependentAndResultsAreOwned(t *testing.T) {
client := &recordingReconciliationClient{responses: []ProposalResponse{ client := &recordingReconciliationClient{responses: []ProposalResponse{
{DuplicateGroups: []DuplicateGroup{ {DuplicateGroups: []DuplicateGroup{
{CandidateIDs: []int{1, 2}, CanonicalCandidateID: 1}, {CandidateNumbers: []int{1, 2}, CanonicalCandidateNumber: 1},
{CandidateIDs: []int{2, 99}, CanonicalCandidateID: 2}, {CandidateNumbers: []int{2, 99}, CanonicalCandidateNumber: 2},
}}, }},
{DuplicateGroups: []DuplicateGroup{}}, {DuplicateGroups: []DuplicateGroup{}},
}} }}
@@ -280,7 +283,7 @@ func cloneProposalResponse(response ProposalResponse) ProposalResponse {
cloned := ProposalResponse{DuplicateGroups: make([]DuplicateGroup, len(response.DuplicateGroups))} cloned := ProposalResponse{DuplicateGroups: make([]DuplicateGroup, len(response.DuplicateGroups))}
for index, group := range response.DuplicateGroups { for index, group := range response.DuplicateGroups {
cloned.DuplicateGroups[index] = DuplicateGroup{ cloned.DuplicateGroups[index] = DuplicateGroup{
CandidateIDs: append([]int(nil), group.CandidateIDs...), CanonicalCandidateID: group.CanonicalCandidateID, CandidateNumbers: append([]int(nil), group.CandidateNumbers...), CanonicalCandidateNumber: group.CanonicalCandidateNumber,
} }
} }
return cloned return cloned

View File

@@ -13,7 +13,7 @@ import (
) )
// Policy identifies the framework-owned reconciliation and assessment rules. // Policy identifies the framework-owned reconciliation and assessment rules.
const Policy = "semantic_reconciliation.v1" const Policy = "semantic_reconciliation.v3"
var _ contracts.ManifestMetadataProvider = (*Engine)(nil) var _ contracts.ManifestMetadataProvider = (*Engine)(nil)
var _ pipeline.CheckpointFingerprintProvider = (*Engine)(nil) var _ pipeline.CheckpointFingerprintProvider = (*Engine)(nil)

View File

@@ -70,16 +70,24 @@ const (
LimitExceeded LimitExceeded
) )
// CandidateMapping relates one model-visible request-local ID to the // CandidateMapping relates one model-visible request-local number to the
// corresponding zero-based position in the caller's candidate slice. // corresponding zero-based position in the caller's candidate slice.
type CandidateMapping struct { type CandidateMapping struct {
CandidateID int CandidateNumber int
CandidatePosition int CandidatePosition int
} }
// CandidateNumberRange is the exact inclusive request-local number range
// presented to the model. Ready preparations always use a one-based range.
type CandidateNumberRange struct {
First int `json:"first"`
Last int `json:"last"`
}
// Preparation owns the visible candidate mapping and prompt materials. // Preparation owns the visible candidate mapping and prompt materials.
type Preparation struct { type Preparation struct {
disposition Disposition disposition Disposition
candidateNumberRange CandidateNumberRange
mappings []CandidateMapping mappings []CandidateMapping
materials contracts.LLMInputSet materials contracts.LLMInputSet
} }
@@ -94,6 +102,12 @@ func (preparation Preparation) CandidateMappings() []CandidateMapping {
return append([]CandidateMapping(nil), preparation.mappings...) return append([]CandidateMapping(nil), preparation.mappings...)
} }
// CandidateNumberRange returns the exact request-local range assigned during
// preparation. A zero value means that no eligible candidates were present.
func (preparation Preparation) CandidateNumberRange() CandidateNumberRange {
return preparation.candidateNumberRange
}
// Materials returns independently owned candidate and transcript materials. // Materials returns independently owned candidate and transcript materials.
// It is empty unless Disposition returns Ready. // It is empty unless Disposition returns Ready.
func (preparation Preparation) Materials() contracts.LLMInputSet { func (preparation Preparation) Materials() contracts.LLMInputSet {
@@ -106,12 +120,13 @@ type sourceRange struct {
} }
type visibleCandidate struct { type visibleCandidate struct {
CandidateID int `json:"candidate_id"` CandidateNumber int `json:"candidate_number"`
Label string `json:"label"` Label string `json:"label"`
SourceRefs []sourceRange `json:"source_refs"` SourceRefs []sourceRange `json:"source_refs"`
} }
type candidateInput struct { type candidateInput struct {
CandidateNumberRange CandidateNumberRange `json:"candidate_number_range"`
Candidates []visibleCandidate `json:"candidates"` Candidates []visibleCandidate `json:"candidates"`
} }
@@ -168,15 +183,18 @@ func Prepare(document *source.SourceDocument, candidates []Candidate, limits Lim
disposition: InsufficientCandidates, disposition: InsufficientCandidates,
mappings: make([]CandidateMapping, len(prepared)), mappings: make([]CandidateMapping, len(prepared)),
} }
if len(prepared) > 0 {
result.candidateNumberRange = CandidateNumberRange{First: 1, Last: len(prepared)}
}
views := make([]visibleCandidate, len(prepared)) views := make([]visibleCandidate, len(prepared))
for index, candidate := range prepared { for index, candidate := range prepared {
candidateID := index + 1 candidateNumber := index + 1
result.mappings[index] = CandidateMapping{ result.mappings[index] = CandidateMapping{
CandidateID: candidateID, CandidateNumber: candidateNumber,
CandidatePosition: candidate.position, CandidatePosition: candidate.position,
} }
views[index] = visibleCandidate{ views[index] = visibleCandidate{
CandidateID: candidateID, CandidateNumber: candidateNumber,
Label: candidates[candidate.position].Label, Label: candidates[candidate.position].Label,
SourceRefs: cloneSourceRanges(candidate.references), SourceRefs: cloneSourceRanges(candidate.references),
} }
@@ -189,7 +207,7 @@ func Prepare(document *source.SourceDocument, candidates []Candidate, limits Lim
return result, nil return result, nil
} }
candidateContent, withinLimit, err := marshalCandidateInput(views, limits.MaximumMaterialBytes) candidateContent, withinLimit, err := marshalCandidateInput(result.candidateNumberRange, views, limits.MaximumMaterialBytes)
if err != nil { if err != nil {
return Preparation{}, fmt.Errorf("prepare semantic reconciliation: encode candidate material: %w", err) return Preparation{}, fmt.Errorf("prepare semantic reconciliation: encode candidate material: %w", err)
} }
@@ -302,10 +320,14 @@ func coalesceIntervals(intervals []sourceInterval) []sourceInterval {
return coalesced return coalesced
} }
func marshalCandidateInput(candidates []visibleCandidate, maximumBytes int) ([]byte, bool, error) { func marshalCandidateInput(numberRange CandidateNumberRange, candidates []visibleCandidate, maximumBytes int) ([]byte, bool, error) {
content := make([]byte, 0, min(maximumBytes, 4096)) content := make([]byte, 0, min(maximumBytes, 4096))
encodedRange, err := json.Marshal(numberRange)
if err != nil {
return nil, false, err
}
var withinLimit bool var withinLimit bool
content, withinLimit = appendWithinLimit(content, maximumBytes, []byte(`{"candidates":[`)) content, withinLimit = appendWithinLimit(content, maximumBytes, []byte(`{"candidate_number_range":`), encodedRange, []byte(`,"candidates":[`))
if !withinLimit { if !withinLimit {
return nil, false, nil return nil, false, nil
} }

View File

@@ -47,11 +47,14 @@ func TestPrepareBuildsContiguousCandidatesAndOwnedSourceContext(t *testing.T) {
t.Fatalf("Prepare() mutated candidates: %#v", candidates) t.Fatalf("Prepare() mutated candidates: %#v", candidates)
} }
if got, want := preparation.CandidateMappings(), []CandidateMapping{ if got, want := preparation.CandidateMappings(), []CandidateMapping{
{CandidateID: 1, CandidatePosition: 0}, {CandidateNumber: 1, CandidatePosition: 0},
{CandidateID: 2, CandidatePosition: 1}, {CandidateNumber: 2, CandidatePosition: 1},
}; !reflect.DeepEqual(got, want) { }; !reflect.DeepEqual(got, want) {
t.Fatalf("CandidateMappings() = %#v, want %#v", got, want) t.Fatalf("CandidateMappings() = %#v, want %#v", got, want)
} }
if got, want := preparation.CandidateNumberRange(), (CandidateNumberRange{First: 1, Last: 2}); got != want {
t.Fatalf("CandidateNumberRange() = %#v, want %#v", got, want)
}
materials := preparation.Materials() materials := preparation.Materials()
if len(materials) != 2 { if len(materials) != 2 {
@@ -79,12 +82,15 @@ func TestPrepareBuildsContiguousCandidatesAndOwnedSourceContext(t *testing.T) {
t.Fatal(err) t.Fatal(err)
} }
wantCandidates := []visibleCandidate{ wantCandidates := []visibleCandidate{
{CandidateID: 1, Label: "The Tavern", SourceRefs: []sourceRange{{StartUnitID: 10, EndUnitID: 20}, {StartUnitID: 90, EndUnitID: 90}}}, {CandidateNumber: 1, Label: "The Tavern", SourceRefs: []sourceRange{{StartUnitID: 10, EndUnitID: 20}, {StartUnitID: 90, EndUnitID: 90}}},
{CandidateID: 2, Label: "The Tavern", SourceRefs: []sourceRange{{StartUnitID: 10, EndUnitID: 20}, {StartUnitID: 90, EndUnitID: 90}}}, {CandidateNumber: 2, Label: "The Tavern", SourceRefs: []sourceRange{{StartUnitID: 10, EndUnitID: 20}, {StartUnitID: 90, EndUnitID: 90}}},
} }
if !reflect.DeepEqual(candidatePayload.Candidates, wantCandidates) { if !reflect.DeepEqual(candidatePayload.Candidates, wantCandidates) {
t.Fatalf("candidate payload = %#v, want %#v", candidatePayload.Candidates, wantCandidates) t.Fatalf("candidate payload = %#v, want %#v", candidatePayload.Candidates, wantCandidates)
} }
if got, want := candidatePayload.CandidateNumberRange, (CandidateNumberRange{First: 1, Last: 2}); got != want {
t.Fatalf("candidate number range = %#v, want %#v", got, want)
}
var candidateObjects struct { var candidateObjects struct {
Candidates []map[string]json.RawMessage `json:"candidates"` Candidates []map[string]json.RawMessage `json:"candidates"`
} }
@@ -92,12 +98,12 @@ func TestPrepareBuildsContiguousCandidatesAndOwnedSourceContext(t *testing.T) {
t.Fatal(err) t.Fatal(err)
} }
for _, candidate := range candidateObjects.Candidates { for _, candidate := range candidateObjects.Candidates {
if len(candidate) != 3 || candidate["candidate_id"] == nil || candidate["label"] == nil || candidate["source_refs"] == nil { if len(candidate) != 3 || candidate["candidate_number"] == nil || candidate["label"] == nil || candidate["source_refs"] == nil {
t.Fatalf("model-facing candidate fields = %#v", candidate) t.Fatalf("model-facing candidate fields = %#v", candidate)
} }
} }
combined := string(materials[candidateInputName].Content) + string(materials[transcriptInputName].Content) combined := string(materials[candidateInputName].Content) + string(materials[transcriptInputName].Content)
for _, forbidden := range []string{document.ID, "application_entity_id", "private-entity-id"} { for _, forbidden := range []string{document.ID, "application_entity_id", "private-entity-id", `"candidate_id"`, `"candidate_ids"`, `"canonical_candidate_id"`} {
if strings.Contains(combined, forbidden) { if strings.Contains(combined, forbidden) {
t.Fatalf("model material leaked %q: %s", forbidden, combined) t.Fatalf("model material leaked %q: %s", forbidden, combined)
} }
@@ -192,11 +198,14 @@ func TestPrepareFiltersUnsafeCandidatesAndCoalescesAdjacentWindows(t *testing.T)
t.Fatalf("Prepare() disposition = %v, error = %v", preparation.Disposition(), err) t.Fatalf("Prepare() disposition = %v, error = %v", preparation.Disposition(), err)
} }
if got, want := preparation.CandidateMappings(), []CandidateMapping{ if got, want := preparation.CandidateMappings(), []CandidateMapping{
{CandidateID: 1, CandidatePosition: 0}, {CandidateNumber: 1, CandidatePosition: 0},
{CandidateID: 2, CandidatePosition: 1}, {CandidateNumber: 2, CandidatePosition: 1},
}; !reflect.DeepEqual(got, want) { }; !reflect.DeepEqual(got, want) {
t.Fatalf("CandidateMappings() = %#v, want %#v", got, want) t.Fatalf("CandidateMappings() = %#v, want %#v", got, want)
} }
if got, want := preparation.CandidateNumberRange(), (CandidateNumberRange{First: 1, Last: 2}); got != want {
t.Fatalf("CandidateNumberRange() after filtering = %#v, want %#v", got, want)
}
var transcript transcriptInput var transcript transcriptInput
if err := json.Unmarshal(preparation.Materials()[transcriptInputName].Content, &transcript); err != nil { if err := json.Unmarshal(preparation.Materials()[transcriptInputName].Content, &transcript); err != nil {
t.Fatal(err) t.Fatal(err)
@@ -208,7 +217,7 @@ func TestPrepareFiltersUnsafeCandidatesAndCoalescesAdjacentWindows(t *testing.T)
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
if got, want := oneCandidate.CandidateMappings(), []CandidateMapping{{CandidateID: 1, CandidatePosition: 0}}; oneCandidate.Disposition() != InsufficientCandidates || !reflect.DeepEqual(got, want) || len(oneCandidate.Materials()) != 0 { if got, want := oneCandidate.CandidateMappings(), []CandidateMapping{{CandidateNumber: 1, CandidatePosition: 0}}; oneCandidate.Disposition() != InsufficientCandidates || !reflect.DeepEqual(got, want) || oneCandidate.CandidateNumberRange() != (CandidateNumberRange{First: 1, Last: 1}) || len(oneCandidate.Materials()) != 0 {
t.Fatalf("Prepare(one candidate) = disposition %v, mappings %#v, materials %#v", oneCandidate.Disposition(), got, oneCandidate.Materials()) t.Fatalf("Prepare(one candidate) = disposition %v, mappings %#v, materials %#v", oneCandidate.Disposition(), got, oneCandidate.Materials())
} }
@@ -216,7 +225,7 @@ func TestPrepareFiltersUnsafeCandidatesAndCoalescesAdjacentWindows(t *testing.T)
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
if nilPreparation.Disposition() != InsufficientCandidates || len(nilPreparation.CandidateMappings()) != 0 || len(nilPreparation.Materials()) != 0 { if nilPreparation.Disposition() != InsufficientCandidates || len(nilPreparation.CandidateMappings()) != 0 || nilPreparation.CandidateNumberRange() != (CandidateNumberRange{}) || len(nilPreparation.Materials()) != 0 {
t.Fatalf("Prepare(nil) = disposition %v, mappings %#v, materials %#v", nilPreparation.Disposition(), nilPreparation.CandidateMappings(), nilPreparation.Materials()) t.Fatalf("Prepare(nil) = disposition %v, mappings %#v, materials %#v", nilPreparation.Disposition(), nilPreparation.CandidateMappings(), nilPreparation.Materials())
} }
} }

View File

@@ -3,6 +3,10 @@ package semanticreconcile
import ( import (
"fmt" "fmt"
"sort" "sort"
"strings"
"unicode/utf8"
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
) )
// ProposalResponse is the complete private structured response contract. // ProposalResponse is the complete private structured response contract.
@@ -10,11 +14,11 @@ type ProposalResponse struct {
DuplicateGroups []DuplicateGroup `json:"duplicate_groups"` DuplicateGroups []DuplicateGroup `json:"duplicate_groups"`
} }
// DuplicateGroup proposes supplied request-local candidate IDs that may denote // DuplicateGroup proposes supplied request-local candidate numbers that may
// one entity and identifies one supplied member as canonical. // denote one entity and identifies one supplied member as canonical.
type DuplicateGroup struct { type DuplicateGroup struct {
CandidateIDs []int `json:"candidate_ids"` CandidateNumbers []int `json:"candidate_numbers"`
CanonicalCandidateID int `json:"canonical_candidate_id"` CanonicalCandidateNumber int `json:"canonical_candidate_number"`
} }
// IssueCategory identifies one stable proposal safety failure. // IssueCategory identifies one stable proposal safety failure.
@@ -31,6 +35,28 @@ const (
IssueOverlappingMember IssueCategory = "overlapping_member" IssueOverlappingMember IssueCategory = "overlapping_member"
) )
var allIssueCategories = []IssueCategory{
IssueMemberNonPositive,
IssueMemberUnknown,
IssueRepeatedMember,
IssueFewerThanTwoMembers,
IssueCanonicalNonPositive,
IssueCanonicalUnknown,
IssueCanonicalNotMember,
IssueOverlappingMember,
}
var issueCorrectionProse = map[IssueCategory]string{
IssueMemberNonPositive: "Use only candidate numbers in %s.",
IssueMemberUnknown: "Remove every candidate number outside %s.",
IssueRepeatedMember: "List each candidate number at most once within the duplicate group.",
IssueFewerThanTwoMembers: "Include at least two distinct candidate numbers, or omit the duplicate group.",
IssueCanonicalNonPositive: "Choose `canonical_candidate_number` from %s.",
IssueCanonicalUnknown: "Choose `canonical_candidate_number` from %s.",
IssueCanonicalNotMember: "Make `canonical_candidate_number` one of the `candidate_numbers` in the same duplicate group.",
IssueOverlappingMember: "Place each candidate number in at most one duplicate group.",
}
// Issue identifies an unsafe proposal category at its original response group // Issue identifies an unsafe proposal category at its original response group
// index without prescribing caller diagnostic text. // index without prescribing caller diagnostic text.
type Issue struct { type Issue struct {
@@ -38,8 +64,8 @@ type Issue struct {
Category IssueCategory Category IssueCategory
} }
// IssueDetails renders stable, domain-neutral proposal diagnostics for an // IssueDetails renders stable, domain-neutral proposal diagnostics for
// adapter's retry message. // operators and debug records. Its internal categories are not model guidance.
func IssueDetails(issues []Issue) []string { func IssueDetails(issues []Issue) []string {
details := make([]string, len(issues)) details := make([]string, len(issues))
for index, issue := range issues { for index, issue := range issues {
@@ -48,6 +74,126 @@ func IssueDetails(issues []Issue) []string {
return details return details
} }
// CorrectionDetails translates proposal issues into stable model-facing prose.
// The response-local group ordinals help the model find the defective group in
// the exact response appended to the correction request.
func CorrectionDetails(issues []Issue, numberRange CandidateNumberRange) ([]string, error) {
if err := validateCorrectionRange(numberRange); err != nil {
return nil, err
}
groupsByCategory := make(map[IssueCategory][]int)
seen := make(map[IssueCategory]map[int]struct{})
for _, issue := range issues {
if issue.GroupIndex < 0 {
return nil, fmt.Errorf("semantic reconciliation issue group index must not be negative")
}
if _, exists := issueCorrectionProse[issue.Category]; !exists {
return nil, fmt.Errorf("semantic reconciliation issue category %q has no correction guidance", issue.Category)
}
if seen[issue.Category] == nil {
seen[issue.Category] = make(map[int]struct{})
}
if _, exists := seen[issue.Category][issue.GroupIndex]; exists {
continue
}
seen[issue.Category][issue.GroupIndex] = struct{}{}
groupsByCategory[issue.Category] = append(groupsByCategory[issue.Category], issue.GroupIndex)
}
details := make([]string, 0, len(groupsByCategory))
includeNamespaceReminder := false
for _, category := range allIssueCategories {
groups := groupsByCategory[category]
if len(groups) == 0 {
continue
}
sort.Ints(groups)
prose := issueCorrectionProse[category]
if issueNeedsCandidateRange(category) {
prose = fmt.Sprintf(prose, candidateRangeDescription(numberRange))
includeNamespaceReminder = true
}
details = append(details, fmt.Sprintf("%s: %s", correctionGroupLabel(groups), prose))
}
if includeNamespaceReminder {
details = append(details, "Candidate numbers are the request-local `candidate_number` values in that range. Transcript unit `id` values and evidence `start_unit_id` and `end_unit_id` values are source positions, not candidate numbers.")
}
return details, nil
}
// CorrectionGuidance builds one bounded request for a complete corrected
// proposal. Additional details let a typed owner append a domain rule without
// weakening or duplicating the shared protocol guidance.
func CorrectionGuidance(issues []Issue, numberRange CandidateNumberRange, additionalDetails ...string) (string, error) {
details, err := CorrectionDetails(issues, numberRange)
if err != nil {
return "", err
}
for _, detail := range additionalDetails {
if !utf8.ValidString(detail) {
return "", fmt.Errorf("semantic reconciliation additional correction detail has invalid UTF-8")
}
detail = strings.TrimSpace(detail)
if detail == "" {
return "", fmt.Errorf("semantic reconciliation additional correction detail must not be blank")
}
details = append(details, detail)
}
if len(details) == 0 {
return "", fmt.Errorf("semantic reconciliation correction guidance requires at least one detail")
}
parts := make([]string, 0, len(details)+2)
parts = append(parts, "The previous semantic-duplicate proposal was invalid.")
parts = append(parts, details...)
parts = append(parts, "Return one complete corrected JSON response that follows the original instructions; do not return a patch or commentary.")
guidance := strings.Join(parts, " ")
if len(guidance) > contracts.MaxNormalizeRetryCorrectionGuidanceBytes {
return "", fmt.Errorf("semantic reconciliation correction guidance exceeds maximum length")
}
return guidance, nil
}
func validateCorrectionRange(numberRange CandidateNumberRange) error {
if numberRange.First != 1 || numberRange.Last < 2 {
return fmt.Errorf("semantic reconciliation correction candidate-number range must start at 1 and include at least two candidates")
}
return nil
}
func candidateRangeDescription(numberRange CandidateNumberRange) string {
return fmt.Sprintf("the inclusive candidate-number range %d through %d", numberRange.First, numberRange.Last)
}
func issueNeedsCandidateRange(category IssueCategory) bool {
switch category {
case IssueMemberNonPositive, IssueMemberUnknown, IssueCanonicalNonPositive, IssueCanonicalUnknown:
return true
default:
return false
}
}
func correctionGroupLabel(groupIndexes []int) string {
const maximumDisplayedGroups = 12
displayed := groupIndexes
if len(displayed) > maximumDisplayedGroups {
displayed = displayed[:maximumDisplayedGroups]
}
ordinals := make([]string, len(displayed))
for index, groupIndex := range displayed {
ordinals[index] = fmt.Sprintf("%d", groupIndex+1)
}
if len(groupIndexes) == 1 {
return "Duplicate group " + ordinals[0]
}
label := "Duplicate groups " + strings.Join(ordinals, ", ")
if omitted := len(groupIndexes) - len(displayed); omitted > 0 {
label += fmt.Sprintf(", and %d additional affected group(s)", omitted)
}
return label
}
// PlanGroup identifies one validated group using original candidate positions. // PlanGroup identifies one validated group using original candidate positions.
type PlanGroup struct { type PlanGroup struct {
memberPositions []int memberPositions []int
@@ -118,18 +264,18 @@ type assessedGroup struct {
conflicting bool conflicting bool
} }
// Assess resolves request-local IDs through the retained preparation mapping // Assess resolves request-local candidate numbers through the retained
// and returns only deterministic, non-overlapping groups. // preparation mapping and returns only deterministic, non-overlapping groups.
func (preparation Preparation) Assess(response ProposalResponse) Assessment { func (preparation Preparation) Assess(response ProposalResponse) Assessment {
positionsByID := make(map[int]int, len(preparation.mappings)) positionsByNumber := make(map[int]int, len(preparation.mappings))
for _, mapping := range preparation.mappings { for _, mapping := range preparation.mappings {
positionsByID[mapping.CandidateID] = mapping.CandidatePosition positionsByNumber[mapping.CandidateNumber] = mapping.CandidatePosition
} }
groups := make([]assessedGroup, len(response.DuplicateGroups)) groups := make([]assessedGroup, len(response.DuplicateGroups))
owners := make(map[int][]int) owners := make(map[int][]int)
for groupIndex, proposal := range response.DuplicateGroups { for groupIndex, proposal := range response.DuplicateGroups {
groups[groupIndex] = assessGroup(proposal, positionsByID) groups[groupIndex] = assessGroup(proposal, positionsByNumber)
if !groups[groupIndex].locallyValid { if !groups[groupIndex].locallyValid {
continue continue
} }
@@ -169,17 +315,17 @@ func (preparation Preparation) Assess(response ProposalResponse) Assessment {
return assessment return assessment
} }
func assessGroup(proposal DuplicateGroup, positionsByID map[int]int) assessedGroup { func assessGroup(proposal DuplicateGroup, positionsByNumber map[int]int) assessedGroup {
group := assessedGroup{} group := assessedGroup{}
seenIDs := make(map[int]struct{}, len(proposal.CandidateIDs)) seenNumbers := make(map[int]struct{}, len(proposal.CandidateNumbers))
memberPositions := make(map[int]struct{}, len(proposal.CandidateIDs)) memberPositions := make(map[int]struct{}, len(proposal.CandidateNumbers))
for _, candidateID := range proposal.CandidateIDs { for _, candidateNumber := range proposal.CandidateNumbers {
if _, repeated := seenIDs[candidateID]; repeated { if _, repeated := seenNumbers[candidateNumber]; repeated {
group.issues = append(group.issues, IssueRepeatedMember) group.issues = append(group.issues, IssueRepeatedMember)
continue continue
} }
seenIDs[candidateID] = struct{}{} seenNumbers[candidateNumber] = struct{}{}
position, category := resolveMember(candidateID, positionsByID) position, category := resolveMember(candidateNumber, positionsByNumber)
if category != "" { if category != "" {
group.issues = append(group.issues, category) group.issues = append(group.issues, category)
continue continue
@@ -191,7 +337,7 @@ func assessGroup(proposal DuplicateGroup, positionsByID map[int]int) assessedGro
group.issues = append(group.issues, IssueFewerThanTwoMembers) group.issues = append(group.issues, IssueFewerThanTwoMembers)
} }
canonicalPosition, canonicalCategory := resolveCanonical(proposal.CanonicalCandidateID, positionsByID) canonicalPosition, canonicalCategory := resolveCanonical(proposal.CanonicalCandidateNumber, positionsByNumber)
if canonicalCategory != "" { if canonicalCategory != "" {
group.issues = append(group.issues, canonicalCategory) group.issues = append(group.issues, canonicalCategory)
} else { } else {
@@ -206,22 +352,22 @@ func assessGroup(proposal DuplicateGroup, positionsByID map[int]int) assessedGro
return group return group
} }
func resolveMember(candidateID int, positionsByID map[int]int) (int, IssueCategory) { func resolveMember(candidateNumber int, positionsByNumber map[int]int) (int, IssueCategory) {
if candidateID <= 0 { if candidateNumber <= 0 {
return 0, IssueMemberNonPositive return 0, IssueMemberNonPositive
} }
position, exists := positionsByID[candidateID] position, exists := positionsByNumber[candidateNumber]
if !exists { if !exists {
return 0, IssueMemberUnknown return 0, IssueMemberUnknown
} }
return position, "" return position, ""
} }
func resolveCanonical(candidateID int, positionsByID map[int]int) (int, IssueCategory) { func resolveCanonical(candidateNumber int, positionsByNumber map[int]int) (int, IssueCategory) {
if candidateID <= 0 { if candidateNumber <= 0 {
return 0, IssueCanonicalNonPositive return 0, IssueCanonicalNonPositive
} }
position, exists := positionsByID[candidateID] position, exists := positionsByNumber[candidateNumber]
if !exists { if !exists {
return 0, IssueCanonicalUnknown return 0, IssueCanonicalUnknown
} }

View File

@@ -2,16 +2,18 @@ package semanticreconcile
import ( import (
"reflect" "reflect"
"strings"
"testing" "testing"
"gitea.maximumdirect.net/eric/notarius/internal/core/source" "gitea.maximumdirect.net/eric/notarius/internal/core/source"
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
) )
func TestAssessProducesAStableOriginalPositionPlan(t *testing.T) { func TestAssessProducesAStableOriginalPositionPlan(t *testing.T) {
preparation := proposalPreparation(t) preparation := proposalPreparation(t)
response := ProposalResponse{DuplicateGroups: []DuplicateGroup{ response := ProposalResponse{DuplicateGroups: []DuplicateGroup{
{CandidateIDs: []int{5, 4}, CanonicalCandidateID: 5}, {CandidateNumbers: []int{5, 4}, CanonicalCandidateNumber: 5},
{CandidateIDs: []int{2, 1}, CanonicalCandidateID: 2}, {CandidateNumbers: []int{2, 1}, CanonicalCandidateNumber: 2},
}} }}
assessment := preparation.Assess(response) assessment := preparation.Assess(response)
want := []planGroupSnapshot{ want := []planGroupSnapshot{
@@ -26,8 +28,8 @@ func TestAssessProducesAStableOriginalPositionPlan(t *testing.T) {
} }
reordered := preparation.Assess(ProposalResponse{DuplicateGroups: []DuplicateGroup{ reordered := preparation.Assess(ProposalResponse{DuplicateGroups: []DuplicateGroup{
{CandidateIDs: []int{1, 2}, CanonicalCandidateID: 2}, {CandidateNumbers: []int{1, 2}, CanonicalCandidateNumber: 2},
{CandidateIDs: []int{4, 5}, CanonicalCandidateID: 5}, {CandidateNumbers: []int{4, 5}, CanonicalCandidateNumber: 5},
}}) }})
if got := snapshotPlan(reordered.Plan()); !reflect.DeepEqual(got, want) { if got := snapshotPlan(reordered.Plan()); !reflect.DeepEqual(got, want) {
t.Fatalf("reordered plan = %#v, want %#v", got, want) t.Fatalf("reordered plan = %#v, want %#v", got, want)
@@ -50,6 +52,79 @@ func TestIssueDetailsPreservesIssueOrder(t *testing.T) {
} }
} }
func TestCorrectionGuidanceCoversEveryIssueCategoryWithoutExposingInternalLabels(t *testing.T) {
numberRange := CandidateNumberRange{First: 1, Last: 8}
if len(issueCorrectionProse) != len(allIssueCategories) {
t.Fatalf("correction prose entries = %d, categories = %d", len(issueCorrectionProse), len(allIssueCategories))
}
issues := make([]Issue, len(allIssueCategories))
seen := make(map[IssueCategory]struct{}, len(allIssueCategories))
for index, category := range allIssueCategories {
if _, duplicate := seen[category]; duplicate {
t.Fatalf("duplicate authoritative issue category %q", category)
}
seen[category] = struct{}{}
if strings.TrimSpace(issueCorrectionProse[category]) == "" {
t.Fatalf("issue category %q has no model-facing prose", category)
}
issues[index] = Issue{GroupIndex: index, Category: category}
}
details, err := CorrectionDetails(issues, numberRange)
if err != nil {
t.Fatalf("CorrectionDetails() error = %v", err)
}
guidance, err := CorrectionGuidance(issues, numberRange)
if err != nil {
t.Fatalf("CorrectionGuidance() error = %v", err)
}
if len(details) != len(allIssueCategories)+1 || !strings.Contains(guidance, "Duplicate group 1") || !strings.Contains(guidance, "candidate-number range 1 through 8") || !strings.Contains(guidance, "start_unit_id") || !strings.Contains(guidance, "complete corrected JSON response") || len(guidance) > contracts.MaxNormalizeRetryCorrectionGuidanceBytes {
t.Fatalf("correction details = %#v guidance = %q", details, guidance)
}
for _, category := range allIssueCategories {
if strings.Contains(guidance, string(category)) {
t.Fatalf("model guidance exposed internal category %q: %q", category, guidance)
}
}
}
func TestCorrectionDetailsDeduplicatesAndBoundsAffectedGroupLists(t *testing.T) {
numberRange := CandidateNumberRange{First: 1, Last: 256}
issues := make([]Issue, 0, 257)
for group := 0; group < 256; group++ {
issues = append(issues, Issue{GroupIndex: group, Category: IssueMemberUnknown})
}
issues = append(issues, Issue{GroupIndex: 0, Category: IssueMemberUnknown})
details, err := CorrectionDetails(issues, numberRange)
if err != nil {
t.Fatalf("CorrectionDetails() error = %v", err)
}
if len(details) != 2 || !strings.Contains(details[0], "additional affected group") || strings.Count(details[0], "Duplicate groups") != 1 || !strings.Contains(details[1], "source positions") {
t.Fatalf("CorrectionDetails() = %#v, want one bounded grouped detail and one namespace reminder", details)
}
guidance, err := CorrectionGuidance(issues, numberRange)
if err != nil || len(guidance) > contracts.MaxNormalizeRetryCorrectionGuidanceBytes {
t.Fatalf("CorrectionGuidance() = %q, %v", guidance, err)
}
}
func TestCorrectionDetailsRejectsUnknownOrInvalidIssues(t *testing.T) {
for _, issues := range [][]Issue{
{{GroupIndex: 0, Category: "future_unmapped_category"}},
{{GroupIndex: -1, Category: IssueMemberUnknown}},
} {
if details, err := CorrectionDetails(issues, CandidateNumberRange{First: 1, Last: 2}); err == nil || details != nil {
t.Fatalf("CorrectionDetails(%#v) = %#v, %v; want fail-closed error", issues, details, err)
}
}
for _, numberRange := range []CandidateNumberRange{{}, {First: 0, Last: 2}, {First: 1, Last: 1}} {
if details, err := CorrectionDetails([]Issue{{GroupIndex: 0, Category: IssueMemberUnknown}}, numberRange); err == nil || details != nil {
t.Fatalf("CorrectionDetails(range %#v) = %#v, %v; want fail-closed error", numberRange, details, err)
}
}
}
func TestAssessRejectsEveryUnsafeLocalGroupShape(t *testing.T) { func TestAssessRejectsEveryUnsafeLocalGroupShape(t *testing.T) {
preparation := proposalPreparation(t) preparation := proposalPreparation(t)
tests := []struct { tests := []struct {
@@ -57,15 +132,15 @@ func TestAssessRejectsEveryUnsafeLocalGroupShape(t *testing.T) {
group DuplicateGroup group DuplicateGroup
category IssueCategory category IssueCategory
}{ }{
{name: "zero member", group: DuplicateGroup{CandidateIDs: []int{0, 2}, CanonicalCandidateID: 2}, category: IssueMemberNonPositive}, {name: "zero member", group: DuplicateGroup{CandidateNumbers: []int{0, 2}, CanonicalCandidateNumber: 2}, category: IssueMemberNonPositive},
{name: "negative member", group: DuplicateGroup{CandidateIDs: []int{-1, 2}, CanonicalCandidateID: 2}, category: IssueMemberNonPositive}, {name: "negative member", group: DuplicateGroup{CandidateNumbers: []int{-1, 2}, CanonicalCandidateNumber: 2}, category: IssueMemberNonPositive},
{name: "unknown member", group: DuplicateGroup{CandidateIDs: []int{99, 2}, CanonicalCandidateID: 2}, category: IssueMemberUnknown}, {name: "unknown member", group: DuplicateGroup{CandidateNumbers: []int{99, 2}, CanonicalCandidateNumber: 2}, category: IssueMemberUnknown},
{name: "repeated member", group: DuplicateGroup{CandidateIDs: []int{1, 1}, CanonicalCandidateID: 1}, category: IssueRepeatedMember}, {name: "repeated member", group: DuplicateGroup{CandidateNumbers: []int{1, 1}, CanonicalCandidateNumber: 1}, category: IssueRepeatedMember},
{name: "too small", group: DuplicateGroup{CandidateIDs: []int{1}, CanonicalCandidateID: 1}, category: IssueFewerThanTwoMembers}, {name: "too small", group: DuplicateGroup{CandidateNumbers: []int{1}, CanonicalCandidateNumber: 1}, category: IssueFewerThanTwoMembers},
{name: "zero canonical", group: DuplicateGroup{CandidateIDs: []int{1, 2}, CanonicalCandidateID: 0}, category: IssueCanonicalNonPositive}, {name: "zero canonical", group: DuplicateGroup{CandidateNumbers: []int{1, 2}, CanonicalCandidateNumber: 0}, category: IssueCanonicalNonPositive},
{name: "negative canonical", group: DuplicateGroup{CandidateIDs: []int{1, 2}, CanonicalCandidateID: -1}, category: IssueCanonicalNonPositive}, {name: "negative canonical", group: DuplicateGroup{CandidateNumbers: []int{1, 2}, CanonicalCandidateNumber: -1}, category: IssueCanonicalNonPositive},
{name: "unknown canonical", group: DuplicateGroup{CandidateIDs: []int{1, 2}, CanonicalCandidateID: 99}, category: IssueCanonicalUnknown}, {name: "unknown canonical", group: DuplicateGroup{CandidateNumbers: []int{1, 2}, CanonicalCandidateNumber: 99}, category: IssueCanonicalUnknown},
{name: "canonical not member", group: DuplicateGroup{CandidateIDs: []int{1, 2}, CanonicalCandidateID: 3}, category: IssueCanonicalNotMember}, {name: "canonical not member", group: DuplicateGroup{CandidateNumbers: []int{1, 2}, CanonicalCandidateNumber: 3}, category: IssueCanonicalNotMember},
} }
for _, test := range tests { for _, test := range tests {
t.Run(test.name, func(t *testing.T) { t.Run(test.name, func(t *testing.T) {
@@ -83,9 +158,9 @@ func TestAssessRejectsEveryUnsafeLocalGroupShape(t *testing.T) {
func TestAssessDiscardsEveryOverlappingGroupAndRetainsIndependentGroups(t *testing.T) { func TestAssessDiscardsEveryOverlappingGroupAndRetainsIndependentGroups(t *testing.T) {
preparation := proposalPreparation(t) preparation := proposalPreparation(t)
assessment := preparation.Assess(ProposalResponse{DuplicateGroups: []DuplicateGroup{ assessment := preparation.Assess(ProposalResponse{DuplicateGroups: []DuplicateGroup{
{CandidateIDs: []int{1, 2}, CanonicalCandidateID: 1}, {CandidateNumbers: []int{1, 2}, CanonicalCandidateNumber: 1},
{CandidateIDs: []int{2, 3}, CanonicalCandidateID: 2}, {CandidateNumbers: []int{2, 3}, CanonicalCandidateNumber: 2},
{CandidateIDs: []int{4, 5}, CanonicalCandidateID: 5}, {CandidateNumbers: []int{4, 5}, CanonicalCandidateNumber: 5},
}}) }})
wantPlan := []planGroupSnapshot{{members: []int{5, 6}, canonical: 6}} wantPlan := []planGroupSnapshot{{members: []int{5, 6}, canonical: 6}}
if got := snapshotPlan(assessment.Plan()); !reflect.DeepEqual(got, wantPlan) { if got := snapshotPlan(assessment.Plan()); !reflect.DeepEqual(got, wantPlan) {
@@ -100,8 +175,8 @@ func TestAssessDiscardsEveryOverlappingGroupAndRetainsIndependentGroups(t *testi
} }
invalidAndSafe := preparation.Assess(ProposalResponse{DuplicateGroups: []DuplicateGroup{ invalidAndSafe := preparation.Assess(ProposalResponse{DuplicateGroups: []DuplicateGroup{
{CandidateIDs: []int{1, 99}, CanonicalCandidateID: 1}, {CandidateNumbers: []int{1, 99}, CanonicalCandidateNumber: 1},
{CandidateIDs: []int{1, 2}, CanonicalCandidateID: 2}, {CandidateNumbers: []int{1, 2}, CanonicalCandidateNumber: 2},
}}) }})
wantPlan = []planGroupSnapshot{{members: []int{0, 2}, canonical: 2}} wantPlan = []planGroupSnapshot{{members: []int{0, 2}, canonical: 2}}
if got := snapshotPlan(invalidAndSafe.Plan()); !reflect.DeepEqual(got, wantPlan) { if got := snapshotPlan(invalidAndSafe.Plan()); !reflect.DeepEqual(got, wantPlan) {
@@ -115,13 +190,13 @@ func TestAssessDiscardsEveryOverlappingGroupAndRetainsIndependentGroups(t *testi
func TestAssessmentAccessorsAndInputsDoNotShareRetainedState(t *testing.T) { func TestAssessmentAccessorsAndInputsDoNotShareRetainedState(t *testing.T) {
preparation := proposalPreparation(t) preparation := proposalPreparation(t)
response := ProposalResponse{DuplicateGroups: []DuplicateGroup{ response := ProposalResponse{DuplicateGroups: []DuplicateGroup{
{CandidateIDs: []int{1, 2}, CanonicalCandidateID: 2}, {CandidateNumbers: []int{1, 2}, CanonicalCandidateNumber: 2},
}} }}
assessment := preparation.Assess(response) assessment := preparation.Assess(response)
want := snapshotPlan(assessment.Plan()) want := snapshotPlan(assessment.Plan())
response.DuplicateGroups[0].CandidateIDs[0] = 99 response.DuplicateGroups[0].CandidateNumbers[0] = 99
response.DuplicateGroups[0].CanonicalCandidateID = 99 response.DuplicateGroups[0].CanonicalCandidateNumber = 99
plan := assessment.Plan() plan := assessment.Plan()
groups := plan.Groups() groups := plan.Groups()
members := groups[0].MemberPositions() members := groups[0].MemberPositions()
@@ -132,7 +207,7 @@ func TestAssessmentAccessorsAndInputsDoNotShareRetainedState(t *testing.T) {
} }
invalid := preparation.Assess(ProposalResponse{DuplicateGroups: []DuplicateGroup{{ invalid := preparation.Assess(ProposalResponse{DuplicateGroups: []DuplicateGroup{{
CandidateIDs: []int{1, 99}, CanonicalCandidateID: 1, CandidateNumbers: []int{1, 99}, CanonicalCandidateNumber: 1,
}}}) }}})
issues := invalid.Issues() issues := invalid.Issues()
issues[0].GroupIndex = 99 issues[0].GroupIndex = 99

View File

@@ -44,20 +44,21 @@ func TestResponseSchemaAcceptsOnlyTheIntegerProposalShape(t *testing.T) {
valid bool valid bool
}{ }{
{name: "empty proposal", value: map[string]any{"duplicate_groups": []any{}}, valid: true}, {name: "empty proposal", value: map[string]any{"duplicate_groups": []any{}}, valid: true},
{name: "valid group", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{1, 2}, "canonical_candidate_id": 1}}}, valid: true}, {name: "valid group", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_numbers": []any{1, 2}, "canonical_candidate_number": 1}}}, valid: true},
{name: "semantic canonical mismatch", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{1, 2}, "canonical_candidate_id": 3}}}, valid: true}, {name: "semantic canonical mismatch", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_numbers": []any{1, 2}, "canonical_candidate_number": 3}}}, valid: true},
{name: "missing proposal", value: map[string]any{}, valid: false}, {name: "missing proposal", value: map[string]any{}, valid: false},
{name: "unknown top-level field", value: map[string]any{"duplicate_groups": []any{}, "extra": true}, valid: false}, {name: "unknown top-level field", value: map[string]any{"duplicate_groups": []any{}, "extra": true}, valid: false},
{name: "missing members", value: map[string]any{"duplicate_groups": []any{map[string]any{"canonical_candidate_id": 1}}}, valid: false}, {name: "missing members", value: map[string]any{"duplicate_groups": []any{map[string]any{"canonical_candidate_number": 1}}}, valid: false},
{name: "missing canonical", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{1, 2}}}}, valid: false}, {name: "missing canonical", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_numbers": []any{1, 2}}}}, valid: false},
{name: "unknown group field", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{1, 2}, "canonical_candidate_id": 1, "name": "replacement"}}}, valid: false}, {name: "unknown group field", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_numbers": []any{1, 2}, "canonical_candidate_number": 1, "name": "replacement"}}}, valid: false},
{name: "too few members", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{1}, "canonical_candidate_id": 1}}}, valid: false}, {name: "former candidate ID fields", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{1, 2}, "canonical_candidate_id": 1}}}, valid: false},
{name: "semantic repeated members", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{1, 1}, "canonical_candidate_id": 1}}}, valid: true}, {name: "too few members", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_numbers": []any{1}, "canonical_candidate_number": 1}}}, valid: false},
{name: "zero member", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{0, 1}, "canonical_candidate_id": 1}}}, valid: false}, {name: "semantic repeated members", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_numbers": []any{1, 1}, "canonical_candidate_number": 1}}}, valid: true},
{name: "negative member", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{-1, 1}, "canonical_candidate_id": 1}}}, valid: false}, {name: "zero member", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_numbers": []any{0, 1}, "canonical_candidate_number": 1}}}, valid: false},
{name: "non-integer member", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{1, 2.5}, "canonical_candidate_id": 1}}}, valid: false}, {name: "negative member", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_numbers": []any{-1, 1}, "canonical_candidate_number": 1}}}, valid: false},
{name: "zero canonical", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{1, 2}, "canonical_candidate_id": 0}}}, valid: false}, {name: "non-integer member", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_numbers": []any{1, 2.5}, "canonical_candidate_number": 1}}}, valid: false},
{name: "contextual selectors", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{1, 2}, "canonical_candidate_id": 1, "source_refs": []any{}}}}, valid: false}, {name: "zero canonical", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_numbers": []any{1, 2}, "canonical_candidate_number": 0}}}, valid: false},
{name: "contextual selectors", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_numbers": []any{1, 2}, "canonical_candidate_number": 1, "source_refs": []any{}}}}, valid: false},
} }
for _, test := range tests { for _, test := range tests {
t.Run(test.name, func(t *testing.T) { t.Run(test.name, func(t *testing.T) {
@@ -72,7 +73,7 @@ func TestResponseSchemaAcceptsOnlyTheIntegerProposalShape(t *testing.T) {
}) })
} }
content := []byte(`{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2}]}`) content := []byte(`{"duplicate_groups":[{"candidate_numbers":[1,2],"canonical_candidate_number":2}]}`)
if err := validateAgainstSchema(content, schema.JSONSchema); err != nil { if err := validateAgainstSchema(content, schema.JSONSchema); err != nil {
t.Fatal(err) t.Fatal(err)
} }
@@ -80,7 +81,7 @@ func TestResponseSchemaAcceptsOnlyTheIntegerProposalShape(t *testing.T) {
if err := json.Unmarshal(content, &response); err != nil { if err := json.Unmarshal(content, &response); err != nil {
t.Fatal(err) t.Fatal(err)
} }
want := ProposalResponse{DuplicateGroups: []DuplicateGroup{{CandidateIDs: []int{1, 2}, CanonicalCandidateID: 2}}} want := ProposalResponse{DuplicateGroups: []DuplicateGroup{{CandidateNumbers: []int{1, 2}, CanonicalCandidateNumber: 2}}}
if !reflect.DeepEqual(response, want) { if !reflect.DeepEqual(response, want) {
t.Fatalf("decoded response = %#v, want %#v", response, want) t.Fatalf("decoded response = %#v, want %#v", response, want)
} }

View File

@@ -45,7 +45,7 @@ func canonicalizeCombatTurn(turn *combatTurnResponse, order shared.SourceRefOrde
if turn == nil { if turn == nil {
return 0, false return 0, false
} }
refs := order.Canonicalize(canonicalSourceRefs(turn.SourceRefs, sourceID)) refs := order.Canonicalize(order.OrderEndpoints(canonicalSourceRefs(turn.SourceRefs, sourceID)))
turn.SourceRefs = combatResponseRefs(refs) turn.SourceRefs = combatResponseRefs(refs)
return order.EarliestValid(refs) return order.EarliestValid(refs)
} }

View File

@@ -15,7 +15,7 @@ import (
const ( const (
Key = "dnd/combat-turns" Key = "dnd/combat-turns"
mappingPolicy = "dnd.combat_turns.extract_mapping.v2" mappingPolicy = "dnd.combat_turns.extract_mapping.v3"
sceneGatePolicy = "dnd.combat_turns.scene_gate.v1" sceneGatePolicy = "dnd.combat_turns.scene_gate.v1"
) )

View File

@@ -25,8 +25,8 @@ func TestExtractMapsAndOrdersCombatTurnsBySourcePosition(t *testing.T) {
}, },
{ {
Actor: "Aria", TurnKind: "reaction", SourceRefs: []combatSourceRefResponse{ Actor: "Aria", TurnKind: "reaction", SourceRefs: []combatSourceRefResponse{
{StartUnitID: 10, EndUnitID: 10}, {StartUnitID: 2, EndUnitID: 10},
{StartUnitID: 10, EndUnitID: 10}, {StartUnitID: 2, EndUnitID: 10},
}, },
}, },
{ {
@@ -42,7 +42,7 @@ func TestExtractMapsAndOrdersCombatTurnsBySourcePosition(t *testing.T) {
if got := []string{result.Value.CombatTurns[0].Actor, result.Value.CombatTurns[1].Actor, result.Value.CombatTurns[2].Actor}; !reflect.DeepEqual(got, []string{"Aria", "Borin", "Unknown"}) { if got := []string{result.Value.CombatTurns[0].Actor, result.Value.CombatTurns[1].Actor, result.Value.CombatTurns[2].Actor}; !reflect.DeepEqual(got, []string{"Aria", "Borin", "Unknown"}) {
t.Fatalf("actor order = %#v, want source-position order with invalid evidence last", got) t.Fatalf("actor order = %#v, want source-position order with invalid evidence last", got)
} }
wantRefs := []source.SourceRef{{SourceID: "session-alpha", StartUnitID: 10, EndUnitID: 10}} wantRefs := []source.SourceRef{{SourceID: "session-alpha", StartUnitID: 10, EndUnitID: 2}}
if !reflect.DeepEqual(result.Value.CombatTurns[0].SourceRefs, wantRefs) { if !reflect.DeepEqual(result.Value.CombatTurns[0].SourceRefs, wantRefs) {
t.Fatalf("canonical refs = %#v, want %#v", result.Value.CombatTurns[0].SourceRefs, wantRefs) t.Fatalf("canonical refs = %#v, want %#v", result.Value.CombatTurns[0].SourceRefs, wantRefs)
} }
@@ -66,7 +66,7 @@ func TestExtractMapsAndOrdersCombatTurnsBySourcePosition(t *testing.T) {
func TestExtractPreservesInvalidCandidatesForValidators(t *testing.T) { func TestExtractPreservesInvalidCandidatesForValidators(t *testing.T) {
client := &fakeCombatTurnsLLMClient{response: extractionResponse{CombatTurns: []combatTurnResponse{ client := &fakeCombatTurnsLLMClient{response: extractionResponse{CombatTurns: []combatTurnResponse{
{ {
Actor: " ", TurnKind: "unsupported", Actor: " ", TurnKind: "turn",
SourceRefs: []combatSourceRefResponse{{StartUnitID: 99, EndUnitID: 0}}, SourceRefs: []combatSourceRefResponse{{StartUnitID: 99, EndUnitID: 0}},
}, },
}}} }}}
@@ -75,7 +75,7 @@ func TestExtractPreservesInvalidCandidatesForValidators(t *testing.T) {
t.Fatalf("Extract() error = %v, want nil for candidate values", err) t.Fatalf("Extract() error = %v, want nil for candidate values", err)
} }
turn := result.Value.CombatTurns[0] turn := result.Value.CombatTurns[0]
if turn.Actor != " " || turn.TurnKind != "unsupported" { if turn.Actor != " " || turn.TurnKind != "turn" {
t.Fatalf("invalid turn fields = %#v, want preserved candidate values", turn) t.Fatalf("invalid turn fields = %#v, want preserved candidate values", turn)
} }
if turn.SourceRefs[0] != (source.SourceRef{SourceID: "session-alpha", StartUnitID: 99}) { if turn.SourceRefs[0] != (source.SourceRef{SourceID: "session-alpha", StartUnitID: 99}) {

View File

@@ -7,6 +7,8 @@ import (
"testing" "testing"
"github.com/santhosh-tekuri/jsonschema/v6" "github.com/santhosh-tekuri/jsonschema/v6"
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
) )
func TestLoadResponseSchemaUsesPrivateCombatShape(t *testing.T) { func TestLoadResponseSchemaUsesPrivateCombatShape(t *testing.T) {
@@ -37,7 +39,7 @@ func TestLoadResponseSchemaUsesPrivateCombatShape(t *testing.T) {
} }
} }
func TestResponseSchemaLeavesSemanticConstraintsToDeterministicValidators(t *testing.T) { func TestResponseSchemaLeavesNonCategoricalSemanticsToDeterministicValidators(t *testing.T) {
schema, err := loadResponseSchema() schema, err := loadResponseSchema()
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
@@ -45,7 +47,6 @@ func TestResponseSchemaLeavesSemanticConstraintsToDeterministicValidators(t *tes
semanticCandidate := validCombatResponse() semanticCandidate := validCombatResponse()
turn := semanticCandidate["combat_turns"].([]any)[0].(map[string]any) turn := semanticCandidate["combat_turns"].([]any)[0].(map[string]any)
turn["actor"] = "" turn["actor"] = ""
turn["turn_kind"] = "unsupported"
ref := turn["source_refs"].([]any)[0].(map[string]any) ref := turn["source_refs"].([]any)[0].(map[string]any)
ref["start_unit_id"] = 0 ref["start_unit_id"] = 0
ref["end_unit_id"] = -1 ref["end_unit_id"] = -1
@@ -54,7 +55,7 @@ func TestResponseSchemaLeavesSemanticConstraintsToDeterministicValidators(t *tes
t.Fatal(err) t.Fatal(err)
} }
if err := validateJSONSchema(content, schema.JSONSchema); err != nil { if err := validateJSONSchema(content, schema.JSONSchema); err != nil {
t.Fatalf("private schema rejected validator-owned semantics: %v", err) t.Fatalf("private schema rejected non-categorical validator-owned semantics: %v", err)
} }
turn["source_refs"] = []any{} turn["source_refs"] = []any{}
content, err = json.Marshal(semanticCandidate) content, err = json.Marshal(semanticCandidate)
@@ -79,6 +80,7 @@ func TestResponseSchemaRetainsStructuralBoundary(t *testing.T) {
{name: "wrong actor type", mutate: func(turn map[string]any) { turn["actor"] = 1 }}, {name: "wrong actor type", mutate: func(turn map[string]any) { turn["actor"] = 1 }},
{name: "unknown field", mutate: func(turn map[string]any) { turn["unexpected"] = true }}, {name: "unknown field", mutate: func(turn map[string]any) { turn["unexpected"] = true }},
{name: "missing source refs", mutate: func(turn map[string]any) { delete(turn, "source_refs") }}, {name: "missing source refs", mutate: func(turn map[string]any) { delete(turn, "source_refs") }},
{name: "unsupported turn kind", mutate: func(turn map[string]any) { turn["turn_kind"] = "unsupported" }},
} { } {
t.Run(test.name, func(t *testing.T) { t.Run(test.name, func(t *testing.T) {
candidate := validCombatResponse() candidate := validCombatResponse()
@@ -94,6 +96,24 @@ func TestResponseSchemaRetainsStructuralBoundary(t *testing.T) {
} }
} }
func TestResponseSchemaAcceptsEverySupportedTurnKind(t *testing.T) {
schema, err := loadResponseSchema()
if err != nil {
t.Fatal(err)
}
for _, kind := range []string{string(dnd.CombatTurnKindTurn), string(dnd.CombatTurnKindReaction), string(dnd.CombatTurnKindLegendaryAction), string(dnd.CombatTurnKindLairAction), string(dnd.CombatTurnKindOther)} {
candidate := validCombatResponse()
candidate["combat_turns"].([]any)[0].(map[string]any)["turn_kind"] = kind
content, err := json.Marshal(candidate)
if err != nil {
t.Fatal(err)
}
if err := validateJSONSchema(content, schema.JSONSchema); err != nil {
t.Fatalf("supported turn kind %q was rejected: %v", kind, err)
}
}
}
func TestResponseSchemaJSONIsMutationSafe(t *testing.T) { func TestResponseSchemaJSONIsMutationSafe(t *testing.T) {
first, err := loadResponseSchema() first, err := loadResponseSchema()
if err != nil { if err != nil {

View File

@@ -13,7 +13,7 @@ import (
const ( const (
Key = "dnd/enemy-events" Key = "dnd/enemy-events"
mappingPolicy = "dnd.enemy_events.extract_mapping.v1" mappingPolicy = "dnd.enemy_events.extract_mapping.v2"
sceneGatePolicy = "dnd.enemy_events.scene_gate.v1" sceneGatePolicy = "dnd.enemy_events.scene_gate.v1"
) )

View File

@@ -16,7 +16,7 @@ import (
func TestExtractMapsEnemyEventsInSourceOrder(t *testing.T) { func TestExtractMapsEnemyEventsInSourceOrder(t *testing.T) {
client := &fakeEnemyEventsLLMClient{response: extractionResponse{Events: []enemyEventResponse{ client := &fakeEnemyEventsLLMClient{response: extractionResponse{Events: []enemyEventResponse{
{Name: "Ashfang", Kind: "killed", SourceRefs: []enemySourceRefResponse{{StartUnitID: 4, EndUnitID: 4}}}, {Name: "Ashfang", Kind: "killed", SourceRefs: []enemySourceRefResponse{{StartUnitID: 4, EndUnitID: 3}}},
{Name: "Ashfang", Kind: "engaged", SourceRefs: []enemySourceRefResponse{{StartUnitID: 1, EndUnitID: 1}, {StartUnitID: 1, EndUnitID: 1}}}, {Name: "Ashfang", Kind: "engaged", SourceRefs: []enemySourceRefResponse{{StartUnitID: 1, EndUnitID: 1}, {StartUnitID: 1, EndUnitID: 1}}},
{Name: "Orcs", Kind: "fled", SourceRefs: []enemySourceRefResponse{{StartUnitID: 3, EndUnitID: 3}}}, {Name: "Orcs", Kind: "fled", SourceRefs: []enemySourceRefResponse{{StartUnitID: 3, EndUnitID: 3}}},
{Name: "One orc", Kind: "captured", SourceRefs: []enemySourceRefResponse{{StartUnitID: 2, EndUnitID: 2}}}, {Name: "One orc", Kind: "captured", SourceRefs: []enemySourceRefResponse{{StartUnitID: 2, EndUnitID: 2}}},
@@ -31,14 +31,17 @@ func TestExtractMapsEnemyEventsInSourceOrder(t *testing.T) {
dnd.EnemyEventKindEngaged, dnd.EnemyEventKindEngaged,
dnd.EnemyEventKindCaptured, dnd.EnemyEventKindCaptured,
dnd.EnemyEventKindIncapacitated, dnd.EnemyEventKindIncapacitated,
dnd.EnemyEventKindFled,
dnd.EnemyEventKindKilled, dnd.EnemyEventKindKilled,
dnd.EnemyEventKindFled,
}) { }) {
t.Fatalf("event order = %#v", got) t.Fatalf("event order = %#v", got)
} }
if refs := result.Value.Events[0].SourceRefs; !reflect.DeepEqual(refs, []source.SourceRef{{SourceID: "combat-session", StartUnitID: 1, EndUnitID: 1}}) { if refs := result.Value.Events[0].SourceRefs; !reflect.DeepEqual(refs, []source.SourceRef{{SourceID: "combat-session", StartUnitID: 1, EndUnitID: 1}}) {
t.Fatalf("canonical source refs = %#v", refs) t.Fatalf("canonical source refs = %#v", refs)
} }
if refs := result.Value.Events[3].SourceRefs; !reflect.DeepEqual(refs, []source.SourceRef{{SourceID: "combat-session", StartUnitID: 3, EndUnitID: 4}}) {
t.Fatalf("reversed source refs = %#v", refs)
}
if len(client.requests) != 1 { if len(client.requests) != 1 {
t.Fatalf("LLM calls = %d, want 1", len(client.requests)) t.Fatalf("LLM calls = %d, want 1", len(client.requests))
} }
@@ -56,14 +59,14 @@ func TestExtractMapsEnemyEventsInSourceOrder(t *testing.T) {
func TestExtractPreservesSemanticCandidatesAndResponseOwnership(t *testing.T) { func TestExtractPreservesSemanticCandidatesAndResponseOwnership(t *testing.T) {
client := &fakeEnemyEventsLLMClient{response: extractionResponse{Events: []enemyEventResponse{{ client := &fakeEnemyEventsLLMClient{response: extractionResponse{Events: []enemyEventResponse{{
Name: " ", Kind: "unsupported", SourceRefs: []enemySourceRefResponse{{StartUnitID: 99, EndUnitID: 0}}, Name: " ", Kind: "engaged", SourceRefs: []enemySourceRefResponse{{StartUnitID: 99, EndUnitID: 0}},
}}}} }}}}
result, err := newEnemyExtractor(t, client).Extract(context.Background(), enemyExtractionRequest(t)) result, err := newEnemyExtractor(t, client).Extract(context.Background(), enemyExtractionRequest(t))
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
event := result.Value.Events[0] event := result.Value.Events[0]
if event.Name != " " || event.Kind != "unsupported" || event.SourceRefs[0] != (source.SourceRef{SourceID: "combat-session", StartUnitID: 99}) { if event.Name != " " || event.Kind != "engaged" || event.SourceRefs[0] != (source.SourceRef{SourceID: "combat-session", StartUnitID: 99}) {
t.Fatalf("semantic candidate = %#v", event) t.Fatalf("semantic candidate = %#v", event)
} }
result.Value.Events[0].SourceRefs[0].StartUnitID = 7 result.Value.Events[0].SourceRefs[0].StartUnitID = 7

View File

@@ -20,7 +20,7 @@ func canonicalEnemyEventList(response extractionResponse, order shared.SourceRef
} }
ordered := make([]orderedEnemyEvent, len(response.Events)) ordered := make([]orderedEnemyEvent, len(response.Events))
for index, event := range response.Events { for index, event := range response.Events {
refs := order.Canonicalize(sourceRefs(event.SourceRefs, sourceID)) refs := order.Canonicalize(order.OrderEndpoints(sourceRefs(event.SourceRefs, sourceID)))
earliest, hasEvidence := order.EarliestValid(refs) earliest, hasEvidence := order.EarliestValid(refs)
ordered[index] = orderedEnemyEvent{ ordered[index] = orderedEnemyEvent{
value: dnd.EnemyEvent{ value: dnd.EnemyEvent{

View File

@@ -7,6 +7,8 @@ import (
"testing" "testing"
"github.com/santhosh-tekuri/jsonschema/v6" "github.com/santhosh-tekuri/jsonschema/v6"
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
) )
func TestResponseSchemaDefinesPrivateStructuralBoundary(t *testing.T) { func TestResponseSchemaDefinesPrivateStructuralBoundary(t *testing.T) {
@@ -28,13 +30,12 @@ func TestResponseSchemaDefinesPrivateStructuralBoundary(t *testing.T) {
semantic := validEnemyResponse() semantic := validEnemyResponse()
event := semantic["events"].([]any)[0].(map[string]any) event := semantic["events"].([]any)[0].(map[string]any)
event["name"] = "" event["name"] = ""
event["kind"] = "unsupported"
ref := event["source_refs"].([]any)[0].(map[string]any) ref := event["source_refs"].([]any)[0].(map[string]any)
ref["start_unit_id"] = 0 ref["start_unit_id"] = 0
ref["end_unit_id"] = -1 ref["end_unit_id"] = -1
content, err = json.Marshal(semantic) content, err = json.Marshal(semantic)
if err != nil || validateEnemySchema(content, schema.JSONSchema) != nil { if err != nil || validateEnemySchema(content, schema.JSONSchema) != nil {
t.Fatalf("validator-owned semantics were rejected: %v", err) t.Fatalf("non-categorical validator-owned semantics were rejected: %v", err)
} }
} }
@@ -46,6 +47,7 @@ func TestResponseSchemaRejectsInvalidStructure(t *testing.T) {
for _, mutate := range []func(map[string]any){ for _, mutate := range []func(map[string]any){
func(event map[string]any) { delete(event, "name") }, func(event map[string]any) { delete(event, "name") },
func(event map[string]any) { event["kind"] = 1 }, func(event map[string]any) { event["kind"] = 1 },
func(event map[string]any) { event["kind"] = "unsupported" },
func(event map[string]any) { event["unexpected"] = true }, func(event map[string]any) { event["unexpected"] = true },
func(event map[string]any) { event["source_refs"].([]any)[0].(map[string]any)["source_id"] = "session" }, func(event map[string]any) { event["source_refs"].([]any)[0].(map[string]any)["source_id"] = "session" },
} { } {
@@ -70,6 +72,24 @@ func TestResponseSchemaRejectsInvalidStructure(t *testing.T) {
} }
} }
func TestResponseSchemaAcceptsEverySupportedKind(t *testing.T) {
schema, err := loadResponseSchema()
if err != nil {
t.Fatal(err)
}
for _, kind := range []string{string(dnd.EnemyEventKindEngaged), string(dnd.EnemyEventKindKilled), string(dnd.EnemyEventKindFled), string(dnd.EnemyEventKindCaptured), string(dnd.EnemyEventKindIncapacitated)} {
candidate := validEnemyResponse()
candidate["events"].([]any)[0].(map[string]any)["kind"] = kind
content, err := json.Marshal(candidate)
if err != nil {
t.Fatal(err)
}
if err := validateEnemySchema(content, schema.JSONSchema); err != nil {
t.Fatalf("supported enemy-event kind %q was rejected: %v", kind, err)
}
}
}
func validEnemyResponse() map[string]any { func validEnemyResponse() map[string]any {
return map[string]any{"events": []any{map[string]any{ return map[string]any{"events": []any{map[string]any{
"name": "Ashfang", "kind": "engaged", "source_refs": []any{map[string]any{"start_unit_id": 1, "end_unit_id": 1}}, "name": "Ashfang", "kind": "engaged", "source_refs": []any{map[string]any{"start_unit_id": 1, "end_unit_id": 1}},

View File

@@ -26,7 +26,7 @@ func canonicalItemOccurrenceList(response extractionResponse, order shared.Sourc
if !found { if !found {
return dnd.ItemOccurrenceList{}, fmt.Errorf("occurrences[%d].name is not in the item registry", index) return dnd.ItemOccurrenceList{}, fmt.Errorf("occurrences[%d].name is not in the item registry", index)
} }
refs := order.Canonicalize(itemOccurrenceSourceRefs(occurrence.SourceRefs, sourceID)) refs := order.Canonicalize(order.OrderEndpoints(itemOccurrenceSourceRefs(occurrence.SourceRefs, sourceID)))
earliest, hasEvidence := order.EarliestValid(refs) earliest, hasEvidence := order.EarliestValid(refs)
ordered[index] = orderedItemOccurrence{ ordered[index] = orderedItemOccurrence{
value: dnd.ItemOccurrence{ value: dnd.ItemOccurrence{

View File

@@ -20,7 +20,7 @@ const (
ItemRegistryMaxBytes = itemregistry.MaxBytes ItemRegistryMaxBytes = itemregistry.MaxBytes
) )
const mappingPolicy = "dnd.item_occurrences.extract_mapping.v2" const mappingPolicy = "dnd.item_occurrences.extract_mapping.v3"
var requiredCapabilities = []string{ var requiredCapabilities = []string{
"chunks", "chunks",

View File

@@ -6,12 +6,14 @@ import (
"testing" "testing"
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts" "gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
itemidentity "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/items/identity" itemidentity "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/items/identity"
itemoccurrenceshape "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/validate/itemoccurrences/shape"
) )
func TestExtractGroundsOccurrencesInRequiredRegistry(t *testing.T) { func TestExtractGroundsOccurrencesInRequiredRegistry(t *testing.T) {
id := itemidentity.DeriveID("Torch") id := itemidentity.DeriveID("Torch")
rawResponse := []byte(`{"occurrences":[{"name":"Torch","kind":"lost","from":"party","source_refs":[{"start_unit_id":1,"end_unit_id":1}]}]}`) rawResponse := []byte(`{"occurrences":[{"name":"Torch","kind":"lost","from":"party","source_refs":[{"start_unit_id":2,"end_unit_id":1}]}]}`)
client := &fakeItemOccurrencesLLMClient{content: append([]byte(nil), rawResponse...)} client := &fakeItemOccurrencesLLMClient{content: append([]byte(nil), rawResponse...)}
req := extractionRequest() req := extractionRequest()
req.References = itemRegistryReferences(t) req.References = itemRegistryReferences(t)
@@ -27,7 +29,7 @@ func TestExtractGroundsOccurrencesInRequiredRegistry(t *testing.T) {
if len(result.Value.Occurrences) != 1 || result.Value.Occurrences[0].ItemID != id || result.Value.Occurrences[0].Name != "Torch" { if len(result.Value.Occurrences) != 1 || result.Value.Occurrences[0].ItemID != id || result.Value.Occurrences[0].Name != "Torch" {
t.Fatalf("occurrences = %#v", result.Value.Occurrences) t.Fatalf("occurrences = %#v", result.Value.Occurrences)
} }
if refs := result.Value.Occurrences[0].SourceRefs; len(refs) != 1 || refs[0].SourceID != req.Source.ID || refs[0].StartUnitID != 1 || refs[0].EndUnitID != 1 { if refs := result.Value.Occurrences[0].SourceRefs; len(refs) != 1 || refs[0].SourceID != req.Source.ID || refs[0].StartUnitID != 1 || refs[0].EndUnitID != 2 {
t.Fatalf("occurrence evidence = %#v, want current-source unit range", refs) t.Fatalf("occurrence evidence = %#v, want current-source unit range", refs)
} }
request := client.requests[0] request := client.requests[0]
@@ -146,3 +148,46 @@ func TestExtractUsesOnlySupportedPromptInputs(t *testing.T) {
t.Fatalf("unexpected prompt input: %#v", client.requests[0].Inputs) t.Fatalf("unexpected prompt input: %#v", client.requests[0].Inputs)
} }
} }
func TestExtractCarriesRejectedHolderGuidanceIntoCorrectedAttempt(t *testing.T) {
references := itemRegistryReferences(t)
req := extractionRequest()
req.References = references
defectiveResponse := []byte(`{"occurrences":[{"name":"Torch","kind":"acquired","quantity":null,"from":"Chest","to":"party","source_refs":[{"start_unit_id":2,"end_unit_id":2}]}]}`)
defective, err := newExtractor(t, &fakeItemOccurrencesLLMClient{content: defectiveResponse}, references).Extract(context.Background(), req)
if err != nil {
t.Fatal(err)
}
validation, err := itemoccurrenceshape.New(itemoccurrenceshape.Options{}).Validate(context.Background(), contracts.TypedValidationRequest[dnd.ItemOccurrenceList]{Value: defective.Value})
if err != nil || validation.Approved {
t.Fatalf("holder validation = %#v, %v; want rejection", validation, err)
}
for _, fragment := range []string{"For `acquired` occurrences", "`from` to JSON null", "item \"Torch\"", "source unit 2"} {
if !strings.Contains(validation.CorrectionGuidance, fragment) {
t.Fatalf("CorrectionGuidance = %q, want %q", validation.CorrectionGuidance, fragment)
}
}
for _, forbidden := range []string{defective.Value.Occurrences[0].ItemID, "item_id", "sha256", itemoccurrenceshape.Key, itemoccurrenceshape.ReasonCode} {
if strings.Contains(validation.CorrectionGuidance, forbidden) {
t.Fatalf("CorrectionGuidance leaked implementation identifier %q: %q", forbidden, validation.CorrectionGuidance)
}
}
correction, err := contracts.NewSemanticCorrection(defective.ModelCandidate.Response, validation.CorrectionGuidance)
if err != nil {
t.Fatal(err)
}
correctedResponse := []byte(`{"occurrences":[{"name":"Torch","kind":"acquired","quantity":null,"from":null,"to":"party","source_refs":[{"start_unit_id":2,"end_unit_id":2}]}]}`)
client := &fakeItemOccurrencesLLMClient{content: correctedResponse}
req.Correction = correction
corrected, err := newExtractor(t, client, references).Extract(context.Background(), req)
if err != nil {
t.Fatal(err)
}
accepted, err := itemoccurrenceshape.New(itemoccurrenceshape.Options{}).Validate(context.Background(), contracts.TypedValidationRequest[dnd.ItemOccurrenceList]{Value: corrected.Value})
if err != nil || !accepted.Approved {
t.Fatalf("corrected holder validation = %#v, %v; want approval", accepted, err)
}
if len(client.requests) != 1 || client.requests[0].Correction == nil || string(client.requests[0].Correction.AssistantResponse) != string(defectiveResponse) || client.requests[0].Correction.UserGuidance != validation.CorrectionGuidance {
t.Fatalf("corrected request = %#v, want byte-faithful defective response and semantic guidance", client.requests)
}
}

View File

@@ -58,6 +58,17 @@ func TestPromptAssetsPrepareItemOccurrencePrompt(t *testing.T) {
t.Fatalf("prepared prompt contains obsolete evidence field %q: %s", obsolete, content) t.Fatalf("prepared prompt contains obsolete evidence field %q: %s", obsolete, content)
} }
} }
for _, holderRule := range []string{
"| `discovered` | `null` | `null` |",
"| `acquired` | `null` | `party` or the named party member gaining possession |",
"| `lost` | `party` or the named party member losing possession | `null` |",
"| `consumed` | `party` or the named party member consuming the item | `null` |",
"| `transferred` | one named party member | a different named party member |",
} {
if !strings.Contains(content, holderRule) {
t.Fatalf("prepared prompt does not include holder rule %q", holderRule)
}
}
} }
func TestPromptAssetsDoNotLeakIntoMetadata(t *testing.T) { func TestPromptAssetsDoNotLeakIntoMetadata(t *testing.T) {

View File

@@ -19,7 +19,7 @@ func TestResponseSchemaIsStrictlyStructuralAndPrivate(t *testing.T) {
} }
valid := map[string]any{"occurrences": []any{ valid := map[string]any{"occurrences": []any{
map[string]any{ map[string]any{
"name": "", "kind": "unsupported", "quantity": 0, "from": "party", "to": "Party", "name": "", "kind": "transferred", "quantity": 0, "from": "party", "to": "Party",
"source_refs": []any{map[string]any{"start_unit_id": 0, "end_unit_id": -1}}, "source_refs": []any{map[string]any{"start_unit_id": 0, "end_unit_id": -1}},
}, },
map[string]any{ map[string]any{
@@ -41,6 +41,7 @@ func TestResponseSchemaIsStrictlyStructuralAndPrivate(t *testing.T) {
{"missing occurrences", map[string]any{}}, {"missing occurrences", map[string]any{}},
{"missing occurrence name", map[string]any{"occurrences": []any{withoutField(responseOccurrence(), "name")}}}, {"missing occurrence name", map[string]any{"occurrences": []any{withoutField(responseOccurrence(), "name")}}},
{"missing nullable field", map[string]any{"occurrences": []any{withoutField(responseOccurrence(), "quantity")}}}, {"missing nullable field", map[string]any{"occurrences": []any{withoutField(responseOccurrence(), "quantity")}}},
{"unsupported kind", map[string]any{"occurrences": []any{withField(responseOccurrence(), "kind", "unsupported")}}},
{"opaque item identifier", map[string]any{"occurrences": []any{withField(responseOccurrence(), "item_id", "item:sha256:opaque")}}}, {"opaque item identifier", map[string]any{"occurrences": []any{withField(responseOccurrence(), "item_id", "item:sha256:opaque")}}},
{"unknown occurrence field", map[string]any{"occurrences": []any{withField(responseOccurrence(), "extra", true)}}}, {"unknown occurrence field", map[string]any{"occurrences": []any{withField(responseOccurrence(), "extra", true)}}},
{"segment-named range", map[string]any{"occurrences": []any{withField(responseOccurrence(), "source_refs", []any{map[string]any{"start_segment": 1, "end_segment": 1}})}}}, {"segment-named range", map[string]any{"occurrences": []any{withField(responseOccurrence(), "source_refs", []any{map[string]any{"start_segment": 1, "end_segment": 1}})}}},

View File

@@ -43,7 +43,7 @@ func canonicalizeItem(item *itemResponse, order shared.SourceRefOrder, sourceID
return 0, false return 0, false
} }
item.Name = identity.NormalizeDisplay(item.Name) item.Name = identity.NormalizeDisplay(item.Name)
refs := order.Canonicalize(canonicalSourceRefs(item.SourceRefs, sourceID)) refs := order.Canonicalize(order.OrderEndpoints(canonicalSourceRefs(item.SourceRefs, sourceID)))
item.SourceRefs = itemResponseRefs(refs) item.SourceRefs = itemResponseRefs(refs)
return order.EarliestValid(refs) return order.EarliestValid(refs)
} }

View File

@@ -14,7 +14,7 @@ import (
const ( const (
Key = "dnd/item-registry" Key = "dnd/item-registry"
mappingPolicy = "dnd.item_registry.extract_mapping.v1" mappingPolicy = "dnd.item_registry.extract_mapping.v2"
) )
var requiredCapabilities = []string{"chunks", "source.transcript"} var requiredCapabilities = []string{"chunks", "source.transcript"}

View File

@@ -15,7 +15,7 @@ import (
func TestExtractMapsItemsWithOwnedEvidenceAndDeterministicOrder(t *testing.T) { func TestExtractMapsItemsWithOwnedEvidenceAndDeterministicOrder(t *testing.T) {
client := &fakeItemsLLMClient{response: extractionResponse{Items: []itemResponse{ client := &fakeItemsLLMClient{response: extractionResponse{Items: []itemResponse{
{Name: " Gold Pieces ", SourceRefs: responseSourceRefs(3, 3)}, {Name: " Gold Pieces ", SourceRefs: responseSourceRefs(3, 2)},
{Name: "Star Compass", SourceRefs: []itemSourceRefResponse{{StartUnitID: 2, EndUnitID: 2}, {StartUnitID: 1, EndUnitID: 1}, {StartUnitID: 1, EndUnitID: 1}}}, {Name: "Star Compass", SourceRefs: []itemSourceRefResponse{{StartUnitID: 2, EndUnitID: 2}, {StartUnitID: 1, EndUnitID: 1}, {StartUnitID: 1, EndUnitID: 1}}},
}}} }}}
result, err := newExtractor(t, client).Extract(context.Background(), extractionRequest()) result, err := newExtractor(t, client).Extract(context.Background(), extractionRequest())
@@ -25,7 +25,7 @@ func TestExtractMapsItemsWithOwnedEvidenceAndDeterministicOrder(t *testing.T) {
refs := []source.SourceRef{{SourceID: "session-items", StartUnitID: 1, EndUnitID: 1}, {SourceID: "session-items", StartUnitID: 2, EndUnitID: 2}} refs := []source.SourceRef{{SourceID: "session-items", StartUnitID: 1, EndUnitID: 1}, {SourceID: "session-items", StartUnitID: 2, EndUnitID: 2}}
want := dnd.ItemRegistry{Items: []dnd.Item{ want := dnd.ItemRegistry{Items: []dnd.Item{
{ID: identity.DeriveID("Star Compass"), Name: "Star Compass", SourceRefs: refs}, {ID: identity.DeriveID("Star Compass"), Name: "Star Compass", SourceRefs: refs},
{ID: identity.DeriveID("Gold Pieces"), Name: "Gold Pieces", SourceRefs: []source.SourceRef{{SourceID: "session-items", StartUnitID: 3, EndUnitID: 3}}}, {ID: identity.DeriveID("Gold Pieces"), Name: "Gold Pieces", SourceRefs: []source.SourceRef{{SourceID: "session-items", StartUnitID: 2, EndUnitID: 3}}},
}} }}
if !reflect.DeepEqual(result.Value, want) { if !reflect.DeepEqual(result.Value, want) {
t.Fatalf("Value = %#v, want %#v", result.Value, want) t.Fatalf("Value = %#v, want %#v", result.Value, want)

View File

@@ -27,7 +27,7 @@ func canonicalOccurrenceList(response extractionResponse, order shared.SourceRef
if !ok { if !ok {
return dnd.LocationOccurrenceList{}, fmt.Errorf("occurrence %d does not match a supplied location selector", index) return dnd.LocationOccurrenceList{}, fmt.Errorf("occurrence %d does not match a supplied location selector", index)
} }
refs := order.Canonicalize(canonicalSourceRefs(occurrence.SourceRefs, sourceID)) refs := order.Canonicalize(order.OrderEndpoints(canonicalSourceRefs(occurrence.SourceRefs, sourceID)))
earliest, hasEvidence := order.EarliestValid(refs) earliest, hasEvidence := order.EarliestValid(refs)
ordered[index] = orderedOccurrence{value: dnd.LocationOccurrence{ ordered[index] = orderedOccurrence{value: dnd.LocationOccurrence{
LocationID: location.ID, LocationID: location.ID,

View File

@@ -15,7 +15,7 @@ import (
const ( const (
Key = "dnd/location-occurrences" Key = "dnd/location-occurrences"
mappingPolicy = "dnd.location_occurrences.extract_mapping.v2" mappingPolicy = "dnd.location_occurrences.extract_mapping.v3"
) )
const ( const (

View File

@@ -23,7 +23,7 @@ func TestExtractMapsKindsOrdersOccurrencesAndPreservesIndependentFacts(t *testin
client := &fakeOccurrencesLLMClient{response: extractionResponse{Occurrences: []occurrenceResponse{ client := &fakeOccurrencesLLMClient{response: extractionResponse{Occurrences: []occurrenceResponse{
{Name: second.Name, RegistryRefs: registryRefs(second), Kind: "mentioned", SourceRefs: occurrenceRefs(30, 30)}, {Name: second.Name, RegistryRefs: registryRefs(second), Kind: "mentioned", SourceRefs: occurrenceRefs(30, 30)},
{Name: first.Name, RegistryRefs: registryRefs(first), Kind: "mentioned", SourceRefs: occurrenceRefs(10, 10)}, {Name: first.Name, RegistryRefs: registryRefs(first), Kind: "mentioned", SourceRefs: occurrenceRefs(10, 10)},
{Name: first.Name, RegistryRefs: registryRefs(first), Kind: "recalled", SourceRefs: occurrenceRefs(10, 10)}, {Name: first.Name, RegistryRefs: registryRefs(first), Kind: "recalled", SourceRefs: occurrenceRefs(20, 10)},
{Name: first.Name, RegistryRefs: registryRefs(first), Kind: "planned", SourceRefs: occurrenceRefs(10, 10)}, {Name: first.Name, RegistryRefs: registryRefs(first), Kind: "planned", SourceRefs: occurrenceRefs(10, 10)},
{Name: first.Name, RegistryRefs: registryRefs(first), Kind: "visited", SourceRefs: append(occurrenceRefs(10, 10), occurrenceRefs(10, 10)...)}, {Name: first.Name, RegistryRefs: registryRefs(first), Kind: "visited", SourceRefs: append(occurrenceRefs(10, 10), occurrenceRefs(10, 10)...)},
{Name: first.Name, RegistryRefs: registryRefs(first), Kind: "visited", SourceRefs: occurrenceRefs(20, 20)}, {Name: first.Name, RegistryRefs: registryRefs(first), Kind: "visited", SourceRefs: occurrenceRefs(20, 20)},
@@ -49,6 +49,9 @@ func TestExtractMapsKindsOrdersOccurrencesAndPreservesIndependentFacts(t *testin
if !reflect.DeepEqual(got[0].SourceRefs, []source.SourceRef{{SourceID: req.Source.ID, StartUnitID: 10, EndUnitID: 10}}) { if !reflect.DeepEqual(got[0].SourceRefs, []source.SourceRef{{SourceID: req.Source.ID, StartUnitID: 10, EndUnitID: 10}}) {
t.Fatalf("canonical evidence = %#v", got[0].SourceRefs) t.Fatalf("canonical evidence = %#v", got[0].SourceRefs)
} }
if !reflect.DeepEqual(got[2].SourceRefs, []source.SourceRef{{SourceID: req.Source.ID, StartUnitID: 10, EndUnitID: 20}}) {
t.Fatalf("reversed evidence = %#v", got[2].SourceRefs)
}
} }
func TestExtractResolvesContextualSelectorsAndUsesCurrentTranscriptEvidenceOnly(t *testing.T) { func TestExtractResolvesContextualSelectorsAndUsesCurrentTranscriptEvidenceOnly(t *testing.T) {

View File

@@ -42,7 +42,7 @@ func canonicalizeLocation(location *locationResponse, order shared.SourceRefOrde
if location == nil { if location == nil {
return 0, false return 0, false
} }
refs := order.Canonicalize(canonicalSourceRefs(location.SourceRefs, sourceID)) refs := order.Canonicalize(order.OrderEndpoints(canonicalSourceRefs(location.SourceRefs, sourceID)))
location.SourceRefs = locationResponseRefs(refs) location.SourceRefs = locationResponseRefs(refs)
return order.EarliestValid(refs) return order.EarliestValid(refs)
} }

View File

@@ -14,7 +14,7 @@ import (
const ( const (
Key = "dnd/location-registry" Key = "dnd/location-registry"
mappingPolicy = "dnd.location_registry.extract_mapping.v1" mappingPolicy = "dnd.location_registry.extract_mapping.v2"
) )
var requiredCapabilities = []string{"chunks", "source.transcript"} var requiredCapabilities = []string{"chunks", "source.transcript"}

View File

@@ -15,7 +15,7 @@ import (
func TestExtractMapsLocationsWithOwnedEvidenceAndDeterministicOrder(t *testing.T) { func TestExtractMapsLocationsWithOwnedEvidenceAndDeterministicOrder(t *testing.T) {
client := &fakeLocationsLLMClient{response: extractionResponse{Locations: []locationResponse{ client := &fakeLocationsLLMClient{response: extractionResponse{Locations: []locationResponse{
{Name: "The Tavern", SourceRefs: responseSourceRefs(3, 3)}, {Name: "The Tavern", SourceRefs: responseSourceRefs(3, 2)},
{Name: "Old Mill", SourceRefs: []locationSourceRefResponse{{StartUnitID: 2, EndUnitID: 2}, {StartUnitID: 1, EndUnitID: 1}, {StartUnitID: 1, EndUnitID: 1}}}, {Name: "Old Mill", SourceRefs: []locationSourceRefResponse{{StartUnitID: 2, EndUnitID: 2}, {StartUnitID: 1, EndUnitID: 1}, {StartUnitID: 1, EndUnitID: 1}}},
}}} }}}
result, err := newExtractor(t, client).Extract(context.Background(), extractionRequest()) result, err := newExtractor(t, client).Extract(context.Background(), extractionRequest())
@@ -25,7 +25,7 @@ func TestExtractMapsLocationsWithOwnedEvidenceAndDeterministicOrder(t *testing.T
refs := []source.SourceRef{{SourceID: "session-locations", StartUnitID: 1, EndUnitID: 1}, {SourceID: "session-locations", StartUnitID: 2, EndUnitID: 2}} refs := []source.SourceRef{{SourceID: "session-locations", StartUnitID: 1, EndUnitID: 1}, {SourceID: "session-locations", StartUnitID: 2, EndUnitID: 2}}
want := dnd.LocationRegistry{Locations: []dnd.Location{ want := dnd.LocationRegistry{Locations: []dnd.Location{
{ID: identity.DeriveID("Old Mill", refs), Name: "Old Mill", SourceRefs: refs}, {ID: identity.DeriveID("Old Mill", refs), Name: "Old Mill", SourceRefs: refs},
{ID: identity.DeriveID("The Tavern", []source.SourceRef{{SourceID: "session-locations", StartUnitID: 3, EndUnitID: 3}}), Name: "The Tavern", SourceRefs: []source.SourceRef{{SourceID: "session-locations", StartUnitID: 3, EndUnitID: 3}}}, {ID: identity.DeriveID("The Tavern", []source.SourceRef{{SourceID: "session-locations", StartUnitID: 2, EndUnitID: 3}}), Name: "The Tavern", SourceRefs: []source.SourceRef{{SourceID: "session-locations", StartUnitID: 2, EndUnitID: 3}}},
}} }}
if !reflect.DeepEqual(result.Value, want) { if !reflect.DeepEqual(result.Value, want) {
t.Fatalf("Value = %#v, want %#v", result.Value, want) t.Fatalf("Value = %#v, want %#v", result.Value, want)

View File

@@ -47,7 +47,7 @@ func canonicalizeOccurrence(occurrence *occurrenceResponse, order shared.SourceR
if occurrence == nil { if occurrence == nil {
return 0, false return 0, false
} }
refs := order.Canonicalize(canonicalSourceRefs(occurrence.SourceRefs, sourceID)) refs := order.Canonicalize(order.OrderEndpoints(canonicalSourceRefs(occurrence.SourceRefs, sourceID)))
occurrence.SourceRefs = occurrenceResponseRefs(refs) occurrence.SourceRefs = occurrenceResponseRefs(refs)
return order.EarliestValid(refs) return order.EarliestValid(refs)
} }

View File

@@ -14,7 +14,7 @@ import (
const ( const (
Key = "dnd/npc-occurrences" Key = "dnd/npc-occurrences"
mappingPolicy = "dnd.npc_occurrences.extract_mapping.v3" mappingPolicy = "dnd.npc_occurrences.extract_mapping.v4"
) )
const ( const (

View File

@@ -22,9 +22,9 @@ func TestExtractMapsEveryKindAndOrdersBySourcePosition(t *testing.T) {
{Name: "Opponent", Kind: "combat_opponent", SourceRefs: occurrenceRefs(20, 20)}, {Name: "Opponent", Kind: "combat_opponent", SourceRefs: occurrenceRefs(20, 20)},
{Name: "Ally", Kind: "combat_ally", SourceRefs: occurrenceRefs(5, 5)}, {Name: "Ally", Kind: "combat_ally", SourceRefs: occurrenceRefs(5, 5)},
{Name: "Speaker", Kind: "dialogue", SourceRefs: append(occurrenceRefs(2, 2), occurrenceRefs(2, 2)...)}, {Name: "Speaker", Kind: "dialogue", SourceRefs: append(occurrenceRefs(2, 2), occurrenceRefs(2, 2)...)},
{Name: "Present", Kind: "noncombat_presence", SourceRefs: occurrenceRefs(7, 7)}, {Name: "Present", Kind: "noncombat_presence", SourceRefs: occurrenceRefs(7, 2)},
{Name: "Mentioned", Kind: "mentioned", SourceRefs: occurrenceRefs(10, 10)}, {Name: "Mentioned", Kind: "mentioned", SourceRefs: occurrenceRefs(10, 10)},
{Name: "Invalid", Kind: "unsupported", SourceRefs: occurrenceRefs(0, 0)}, {Name: "Invalid", Kind: "other", SourceRefs: occurrenceRefs(0, 0)},
}}} }}}
references := requiredRegistryReferences(t, "Mentioned", "Speaker", "Present", "Ally", "Opponent", "Other", "Invalid") references := requiredRegistryReferences(t, "Mentioned", "Speaker", "Present", "Ally", "Opponent", "Other", "Invalid")
req := extractionRequest() req := extractionRequest()
@@ -44,17 +44,20 @@ func TestExtractMapsEveryKindAndOrdersBySourcePosition(t *testing.T) {
dnd.NPCOccurrenceKindCombatAlly, dnd.NPCOccurrenceKindCombatAlly,
dnd.NPCOccurrenceKindCombatOpponent, dnd.NPCOccurrenceKindCombatOpponent,
dnd.NPCOccurrenceKindOther, dnd.NPCOccurrenceKindOther,
"unsupported", dnd.NPCOccurrenceKindOther,
}) { }) {
t.Fatalf("occurrence kinds = %#v", got) t.Fatalf("occurrence kinds = %#v", got)
} }
if refs := result.Value.Occurrences[1].SourceRefs; !reflect.DeepEqual(refs, []source.SourceRef{{SourceID: "session-alpha", StartUnitID: 2, EndUnitID: 2}}) { if refs := result.Value.Occurrences[1].SourceRefs; !reflect.DeepEqual(refs, []source.SourceRef{{SourceID: "session-alpha", StartUnitID: 2, EndUnitID: 2}}) {
t.Fatalf("canonical source refs = %#v", refs) t.Fatalf("canonical source refs = %#v", refs)
} }
if refs := result.Value.Occurrences[2].SourceRefs; !reflect.DeepEqual(refs, []source.SourceRef{{SourceID: "session-alpha", StartUnitID: 2, EndUnitID: 7}}) {
t.Fatalf("reversed source refs = %#v", refs)
}
if id := result.Value.Occurrences[0].NPCID; id != identity.DeriveID("Mentioned") { if id := result.Value.Occurrences[0].NPCID; id != identity.DeriveID("Mentioned") {
t.Fatalf("durable NPC ID = %q, want registry identity", id) t.Fatalf("durable NPC ID = %q, want registry identity", id)
} }
if invalid := result.Value.Occurrences[6]; invalid.Name != "Invalid" || invalid.Kind != "unsupported" || !reflect.DeepEqual(invalid.SourceRefs, []source.SourceRef{{SourceID: "session-alpha"}}) { if invalid := result.Value.Occurrences[6]; invalid.Name != "Invalid" || invalid.Kind != dnd.NPCOccurrenceKindOther || !reflect.DeepEqual(invalid.SourceRefs, []source.SourceRef{{SourceID: "session-alpha"}}) {
t.Fatalf("invalid candidate = %#v, want preserved values with current source identity", invalid) t.Fatalf("invalid candidate = %#v, want preserved values with current source identity", invalid)
} }
if len(client.requests) != 1 { if len(client.requests) != 1 {

View File

@@ -7,6 +7,8 @@ import (
"testing" "testing"
"github.com/santhosh-tekuri/jsonschema/v6" "github.com/santhosh-tekuri/jsonschema/v6"
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
) )
func TestResponseSchemaOwnsOnlyPrivateStructuralContract(t *testing.T) { func TestResponseSchemaOwnsOnlyPrivateStructuralContract(t *testing.T) {
@@ -29,7 +31,6 @@ func TestResponseSchemaOwnsOnlyPrivateStructuralContract(t *testing.T) {
semanticCandidate := validOccurrenceResponse() semanticCandidate := validOccurrenceResponse()
occurrence := semanticCandidate["occurrences"].([]any)[0].(map[string]any) occurrence := semanticCandidate["occurrences"].([]any)[0].(map[string]any)
occurrence["name"] = "" occurrence["name"] = ""
occurrence["kind"] = "unsupported"
ref := occurrence["source_refs"].([]any)[0].(map[string]any) ref := occurrence["source_refs"].([]any)[0].(map[string]any)
ref["start_unit_id"] = 0 ref["start_unit_id"] = 0
ref["end_unit_id"] = -1 ref["end_unit_id"] = -1
@@ -38,12 +39,13 @@ func TestResponseSchemaOwnsOnlyPrivateStructuralContract(t *testing.T) {
t.Fatal(err) t.Fatal(err)
} }
if err := validateJSONSchema(content, schema.JSONSchema); err != nil { if err := validateJSONSchema(content, schema.JSONSchema); err != nil {
t.Fatalf("schema rejected validator-owned semantics: %v", err) t.Fatalf("schema rejected non-categorical validator-owned semantics: %v", err)
} }
for _, mutate := range []func(map[string]any){ for _, mutate := range []func(map[string]any){
func(record map[string]any) { delete(record, "name") }, func(record map[string]any) { delete(record, "name") },
func(record map[string]any) { record["kind"] = 1 }, func(record map[string]any) { record["kind"] = 1 },
func(record map[string]any) { record["kind"] = "unsupported" },
func(record map[string]any) { record["npc_id"] = "npc:sha256:opaque" }, func(record map[string]any) { record["npc_id"] = "npc:sha256:opaque" },
func(record map[string]any) { record["unexpected"] = true }, func(record map[string]any) { record["unexpected"] = true },
func(record map[string]any) { func(record map[string]any) {
@@ -79,6 +81,24 @@ func TestResponseSchemaIsDefensiveAndContentSafe(t *testing.T) {
} }
} }
func TestResponseSchemaAcceptsEverySupportedKind(t *testing.T) {
schema, err := loadResponseSchema()
if err != nil {
t.Fatal(err)
}
for _, kind := range []string{string(dnd.NPCOccurrenceKindMentioned), string(dnd.NPCOccurrenceKindNoncombatPresence), string(dnd.NPCOccurrenceKindDialogue), string(dnd.NPCOccurrenceKindCombatAlly), string(dnd.NPCOccurrenceKindCombatOpponent), string(dnd.NPCOccurrenceKindOther)} {
candidate := validOccurrenceResponse()
candidate["occurrences"].([]any)[0].(map[string]any)["kind"] = kind
content, err := json.Marshal(candidate)
if err != nil {
t.Fatal(err)
}
if err := validateJSONSchema(content, schema.JSONSchema); err != nil {
t.Fatalf("supported NPC-occurrence kind %q was rejected: %v", kind, err)
}
}
}
func validOccurrenceResponse() map[string]any { func validOccurrenceResponse() map[string]any {
return map[string]any{"occurrences": []any{map[string]any{ return map[string]any{"occurrences": []any{map[string]any{
"name": "Mira Thorn", "kind": "dialogue", "name": "Mira Thorn", "kind": "dialogue",

View File

@@ -46,7 +46,7 @@ func canonicalizeNPC(npc *npcResponse, order shared.SourceRefOrder, sourceID str
if npc == nil { if npc == nil {
return 0, false return 0, false
} }
refs := order.Canonicalize(canonicalSourceRefs(npc.SourceRefs, sourceID)) refs := order.Canonicalize(order.OrderEndpoints(canonicalSourceRefs(npc.SourceRefs, sourceID)))
npc.SourceRefs = npcResponseRefs(refs) npc.SourceRefs = npcResponseRefs(refs)
return order.EarliestValid(refs) return order.EarliestValid(refs)
} }

View File

@@ -13,7 +13,7 @@ import (
const Key = "dnd/npc-registry" const Key = "dnd/npc-registry"
const mappingPolicy = "dnd.npc_registry.extract_mapping.v2" const mappingPolicy = "dnd.npc_registry.extract_mapping.v3"
var requiredCapabilities = []string{ var requiredCapabilities = []string{
"chunks", "chunks",

View File

@@ -16,7 +16,7 @@ import (
func TestExtractReturnsCanonicalNPCRegistryFromPrivateResponse(t *testing.T) { func TestExtractReturnsCanonicalNPCRegistryFromPrivateResponse(t *testing.T) {
client := &fakeNPCsLLMClient{response: extractionResponse{NPCs: []npcResponse{ client := &fakeNPCsLLMClient{response: extractionResponse{NPCs: []npcResponse{
{ {
Name: "Captain Vale", SourceRefs: responseSourceRefs(3, 3), Name: "Captain Vale", SourceRefs: responseSourceRefs(3, 2),
}, },
{ {
Name: "Mira Thorn", Name: "Mira Thorn",
@@ -34,7 +34,7 @@ func TestExtractReturnsCanonicalNPCRegistryFromPrivateResponse(t *testing.T) {
} }
want := dnd.NPCRegistry{NPCs: []dnd.NPC{ want := dnd.NPCRegistry{NPCs: []dnd.NPC{
{ID: identity.DeriveID("Mira Thorn"), Name: "Mira Thorn", SourceRefs: []source.SourceRef{{SourceID: "session-alpha", StartUnitID: 1, EndUnitID: 2}, {SourceID: "session-alpha", StartUnitID: 2, EndUnitID: 2}}}, {ID: identity.DeriveID("Mira Thorn"), Name: "Mira Thorn", SourceRefs: []source.SourceRef{{SourceID: "session-alpha", StartUnitID: 1, EndUnitID: 2}, {SourceID: "session-alpha", StartUnitID: 2, EndUnitID: 2}}},
{ID: identity.DeriveID("Captain Vale"), Name: "Captain Vale", SourceRefs: []source.SourceRef{{SourceID: "session-alpha", StartUnitID: 3, EndUnitID: 3}}}, {ID: identity.DeriveID("Captain Vale"), Name: "Captain Vale", SourceRefs: []source.SourceRef{{SourceID: "session-alpha", StartUnitID: 2, EndUnitID: 3}}},
}} }}
if !reflect.DeepEqual(result.Value, want) { if !reflect.DeepEqual(result.Value, want) {
t.Fatalf("Value = %#v, want %#v", result.Value, want) t.Fatalf("Value = %#v, want %#v", result.Value, want)

View File

@@ -75,18 +75,18 @@ func TestExtractReturnsSemanticallyInvalidResponseForDeterministicValidation(t *
if err != nil { if err != nil {
t.Fatalf("loadResponseSchema() error = %v", err) t.Fatalf("loadResponseSchema() error = %v", err)
} }
if err := validateJSONSchema(t, map[string]any{"kind": "unrecognized", "title": " ", "summary": ""}, schema.JSONSchema); err != nil { if err := validateJSONSchema(t, map[string]any{"kind": "narrative", "title": " ", "summary": ""}, schema.JSONSchema); err != nil {
t.Fatalf("semantic candidate rejected by private schema: %v", err) t.Fatalf("semantic candidate rejected by private schema: %v", err)
} }
client := &fakeSceneDescriptionsLLMClient{response: extractionResponse{ client := &fakeSceneDescriptionsLLMClient{response: extractionResponse{
Kind: dnd.SceneKind("unrecognized"), Title: " ", Summary: "", Kind: dnd.SceneKindNarrative, Title: " ", Summary: "",
}} }}
result, err := newExtractor(t, client).Extract(context.Background(), extractionRequest()) result, err := newExtractor(t, client).Extract(context.Background(), extractionRequest())
if err != nil { if err != nil {
t.Fatalf("Extract() error = %v, want nil", err) t.Fatalf("Extract() error = %v, want nil", err)
} }
scene := result.Value.Scenes[0] scene := result.Value.Scenes[0]
if scene.Kind != dnd.SceneKind("unrecognized") || scene.Title != "" || scene.Summary != "" { if scene.Kind != dnd.SceneKindNarrative || scene.Title != "" || scene.Summary != "" {
t.Fatalf("scene = %#v, want semantic candidates returned for deterministic validation", scene) t.Fatalf("scene = %#v, want semantic candidates returned for deterministic validation", scene)
} }
} }

View File

@@ -24,6 +24,7 @@ var promptAssetManifest = shared.PromptAssetManifest{
"common-dnd-identity.md", "common-dnd-identity.md",
"common-dnd-references.md", "common-dnd-references.md",
"common-dnd-transcript-chunk.md", "common-dnd-transcript-chunk.md",
"common-dnd-scene-combat-policy.md",
}, },
} }

View File

@@ -42,6 +42,23 @@ func TestRegisterPromptAssetsPreparesSceneDescriptionPrompt(t *testing.T) {
if prepared.PromptID != PromptID || prepared.OutputContract.SchemaPath != "dnd_scene_descriptions_llm.v1.json" { if prepared.PromptID != PromptID || prepared.OutputContract.SchemaPath != "dnd_scene_descriptions_llm.v1.json" {
t.Fatalf("prepared prompt = %#v, want scene-description prompt identity and schema wiring", prepared) t.Fatalf("prepared prompt = %#v, want scene-description prompt identity and schema wiring", prepared)
} }
const combatPolicySentinel = "substantive active combat materially organizes the"
transcriptIndex := -1
policyIndex := -1
instructionsIndex := len(prepared.Messages) - 1
policyOccurrences := 0
for index, message := range prepared.Messages {
if strings.Contains(message.Content, "scene-description-transcript") {
transcriptIndex = index
}
if count := strings.Count(message.Content, combatPolicySentinel); count > 0 {
policyIndex = index
policyOccurrences += count
}
}
if policyOccurrences != 1 || policyIndex <= transcriptIndex || policyIndex >= instructionsIndex {
t.Fatalf("combat policy appeared %d times at message %d; want once after transcript message %d and before instructions message %d", policyOccurrences, policyIndex, transcriptIndex, instructionsIndex)
}
} }
func TestPromptMetadataAndDiagnosticsDoNotContainRawAssets(t *testing.T) { func TestPromptMetadataAndDiagnosticsDoNotContainRawAssets(t *testing.T) {

View File

@@ -7,6 +7,8 @@ import (
"testing" "testing"
"github.com/santhosh-tekuri/jsonschema/v6" "github.com/santhosh-tekuri/jsonschema/v6"
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
) )
func TestLoadResponseSchemaUsesStrictPrivateSceneDescriptionContract(t *testing.T) { func TestLoadResponseSchemaUsesStrictPrivateSceneDescriptionContract(t *testing.T) {
@@ -30,7 +32,7 @@ func TestLoadResponseSchemaUsesStrictPrivateSceneDescriptionContract(t *testing.
{name: "unknown framework field", response: map[string]any{"kind": "combat", "title": "Ambush", "summary": "Bandits strike.", "id": "assigned-later"}}, {name: "unknown framework field", response: map[string]any{"kind": "combat", "title": "Ambush", "summary": "Bandits strike.", "id": "assigned-later"}},
{name: "unknown application field", response: map[string]any{"kind": "combat", "title": "Ambush", "summary": "Bandits strike.", "source_ref": map[string]any{}}}, {name: "unknown application field", response: map[string]any{"kind": "combat", "title": "Ambush", "summary": "Bandits strike.", "source_ref": map[string]any{}}},
{name: "collection is not allowed", response: map[string]any{"kind": "combat", "title": "Ambush", "summary": "Bandits strike.", "scenes": []any{}}}, {name: "collection is not allowed", response: map[string]any{"kind": "combat", "title": "Ambush", "summary": "Bandits strike.", "scenes": []any{}}},
{name: "unsupported kind", response: map[string]any{"kind": "interlude", "title": "Ambush", "summary": "Bandits strike."}, valid: true}, {name: "unsupported kind", response: map[string]any{"kind": "interlude", "title": "Ambush", "summary": "Bandits strike."}},
{name: "empty title", response: map[string]any{"kind": "combat", "title": "", "summary": "Bandits strike."}, valid: true}, {name: "empty title", response: map[string]any{"kind": "combat", "title": "", "summary": "Bandits strike."}, valid: true},
{name: "empty summary", response: map[string]any{"kind": "combat", "title": "Ambush", "summary": ""}, valid: true}, {name: "empty summary", response: map[string]any{"kind": "combat", "title": "Ambush", "summary": ""}, valid: true},
{name: "wrong kind type", response: map[string]any{"kind": 7, "title": "Ambush", "summary": "Bandits strike."}}, {name: "wrong kind type", response: map[string]any{"kind": 7, "title": "Ambush", "summary": "Bandits strike."}},
@@ -64,6 +66,19 @@ func TestResponseSchemaIsMutationSafeAndDiagnosticsRedactContent(t *testing.T) {
} }
} }
func TestResponseSchemaAcceptsEverySupportedKind(t *testing.T) {
schema, err := loadResponseSchema()
if err != nil {
t.Fatal(err)
}
for _, kind := range []string{string(dnd.SceneKindCombat), string(dnd.SceneKindNarrative), string(dnd.SceneKindRecap), string(dnd.SceneKindMeta)} {
response := map[string]any{"kind": kind, "title": "Title", "summary": "Summary"}
if err := validateJSONSchema(t, response, schema.JSONSchema); err != nil {
t.Fatalf("supported scene kind %q was rejected: %v", kind, err)
}
}
}
func validateJSONSchema(t *testing.T, instance map[string]any, schemaContent []byte) error { func validateJSONSchema(t *testing.T, instance map[string]any, schemaContent []byte) error {
t.Helper() t.Helper()
content, err := json.Marshal(instance) content, err := json.Marshal(instance)

View File

@@ -45,7 +45,7 @@ func canonicalizeSpellCast(spell *spellCastResponse, order shared.SourceRefOrder
if spell == nil { if spell == nil {
return 0, false return 0, false
} }
refs := order.Canonicalize(spellSourceRefs(spell.SourceRefs, sourceID)) refs := order.Canonicalize(order.OrderEndpoints(spellSourceRefs(spell.SourceRefs, sourceID)))
spell.SourceRefs = spellResponseRefs(refs) spell.SourceRefs = spellResponseRefs(refs)
return order.EarliestValid(refs) return order.EarliestValid(refs)
} }

View File

@@ -16,7 +16,7 @@ import (
const Key = "dnd/spells" const Key = "dnd/spells"
const SchemaVersion = "v1" const SchemaVersion = "v1"
const mappingPolicy = "dnd.spells.extract_mapping.v2" const mappingPolicy = "dnd.spells.extract_mapping.v3"
const ( const (
NPCRegistryReferenceSlot = npcregistry.ReferenceSlot NPCRegistryReferenceSlot = npcregistry.ReferenceSlot

View File

@@ -21,7 +21,7 @@ func TestExtractReturnsCanonicalSpellListFromPrivateResponse(t *testing.T) {
{ {
Caster: " Aria ", Caster: " Aria ",
Spell: " Cure Wounds ", Spell: " Cure Wounds ",
SourceRefs: responseSourceRefs(1, 2), SourceRefs: responseSourceRefs(2, 1),
}, },
}}} }}}
req := extractionRequest() req := extractionRequest()

View File

@@ -23,7 +23,7 @@ const (
Key = "dnd/item-registry" Key = "dnd/item-registry"
PromptID = "dnd.item_registry.normalize" PromptID = "dnd.item_registry.normalize"
PromptVersion = "v1" PromptVersion = "v1"
normalizationPolicy = "dnd.item_registry.normalize.v3" normalizationPolicy = "dnd.item_registry.normalize.v4"
NormalizationPolicy = normalizationPolicy NormalizationPolicy = normalizationPolicy
ReasonCodeItemFieldsNormalized = "item_fields_normalized" ReasonCodeItemFieldsNormalized = "item_fields_normalized"
@@ -129,7 +129,7 @@ func (n *Normalizer) Normalize(ctx context.Context, req contracts.TypedNormalize
case semanticreconcile.SkippedInsufficientCandidates: case semanticreconcile.SkippedInsufficientCandidates:
return normalizationResult(deterministic, findings, nil, nil) return normalizationResult(deterministic, findings, nil, nil)
case semanticreconcile.SkippedLimitExceeded: case semanticreconcile.SkippedLimitExceeded:
return fallbackResult(deterministic, findings, nil, semanticFallbackFinding(-1)) return fallbackResult(deterministic, findings, nil, semanticFallbackFinding(-1, "the candidate set exceeded the configured semantic-reconciliation limits"))
case semanticreconcile.RetryableInvalidStructuredOutput: case semanticreconcile.RetryableInvalidStructuredOutput:
return n.invalidStructuredResult(deterministic, findings) return n.invalidStructuredResult(deterministic, findings)
case semanticreconcile.Complete, semanticreconcile.RetryableDiscardedProposalGroups: case semanticreconcile.Complete, semanticreconcile.RetryableDiscardedProposalGroups:
@@ -150,22 +150,38 @@ func (n *Normalizer) Normalize(ctx context.Context, req contracts.TypedNormalize
} }
func (n *Normalizer) invalidStructuredResult(value dnd.ItemRegistry, findings []diagnostics.Finding) (contracts.TypedNormalizeResult[dnd.ItemRegistry], error) { func (n *Normalizer) invalidStructuredResult(value dnd.ItemRegistry, findings []diagnostics.Finding) (contracts.TypedNormalizeResult[dnd.ItemRegistry], error) {
return retryResultWithFallback(value, findings, nil, nil, ReasonCodeItemSemanticRetryProposalInvalid, "semantic proposal requires retry: invalid structured output", semanticFallbackFinding(-1)) return retryResultWithFallback(value, findings, nil, nil, ReasonCodeItemSemanticRetryProposalInvalid, "", "semantic proposal requires retry: invalid structured output", semanticFallbackFinding(-1, "the final model response did not satisfy the structured-output contract"))
} }
func retryResult(value dnd.ItemRegistry, findings, advisoryFindings []diagnostics.Finding, reconciliation semanticreconcile.Result, rejectedGroups int) (contracts.TypedNormalizeResult[dnd.ItemRegistry], error) { func retryResult(value dnd.ItemRegistry, findings, advisoryFindings []diagnostics.Finding, reconciliation semanticreconcile.Result, rejectedGroups int) (contracts.TypedNormalizeResult[dnd.ItemRegistry], error) {
details := semanticreconcile.IssueDetails(reconciliation.Issues()) issues := reconciliation.Issues()
operatorDetails := semanticreconcile.IssueDetails(issues)
correctionDetails, err := reconciliation.CorrectionDetails()
if err != nil {
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{}, normalizerErrorf("build semantic correction details: %w", err)
}
var additionalGuidance []string
if rejectedGroups > 0 { if rejectedGroups > 0 {
details = append(details, "currency may only be consolidated with aliases of one denomination") const currencyGuidance = "Consolidate currency only with aliases of the same denomination; keep different denominations separate and do not combine currency with non-currency items."
operatorDetails = append(operatorDetails, "currency may only be consolidated with aliases of one denomination")
correctionDetails = append(correctionDetails, currencyGuidance)
additionalGuidance = append(additionalGuidance, currencyGuidance)
}
correctionGuidance, err := reconciliation.CorrectionGuidance(additionalGuidance...)
if err != nil {
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{}, normalizerErrorf("build semantic correction guidance: %w", err)
} }
discardedGroups := reconciliation.DiscardedGroupCount() + rejectedGroups discardedGroups := reconciliation.DiscardedGroupCount() + rejectedGroups
return retryResultWithFallback(value, findings, advisoryFindings, reconciliation.ModelCandidate(), ReasonCodeItemSemanticRetryProposalInvalid, diagnostics.Aggregate("semantic proposal requires retry", details), semanticFallbackFinding(discardedGroups)) return retryResultWithFallback(value, findings, advisoryFindings, reconciliation.ModelCandidate(), ReasonCodeItemSemanticRetryProposalInvalid, correctionGuidance, diagnostics.Aggregate("semantic proposal requires retry", operatorDetails), semanticFallbackFinding(discardedGroups, correctionDetails...))
} }
func semanticFallbackFinding(discarded int) diagnostics.Finding { func semanticFallbackFinding(discarded int, details ...string) diagnostics.Finding {
message := "semantic proposal could not be applied" message := "semantic reconciliation could not be completed; all affected candidates were preserved separately"
if discarded >= 0 { if discarded >= 0 {
message = fmt.Sprintf("%d proposal group(s) omitted after semantic proposal retry exhaustion", discarded) message = fmt.Sprintf("%d proposal group(s) omitted from the final semantic proposal; all affected candidates were preserved separately", discarded)
}
if len(details) > 0 {
message = diagnostics.Aggregate(message, details)
} }
return diagnostics.Finding{Scope: "items", ReasonCode: ReasonCodeItemSemanticReconciliationExhausted, Message: message} return diagnostics.Finding{Scope: "items", ReasonCode: ReasonCodeItemSemanticReconciliationExhausted, Message: message}
} }
@@ -196,7 +212,7 @@ func fallbackResult(value dnd.ItemRegistry, findings, advisoryFindings []diagnos
return result, nil return result, nil
} }
func retryResultWithFallback(value dnd.ItemRegistry, findings, advisoryFindings []diagnostics.Finding, candidate *contracts.ModelCandidate, reasonCode, message string, fallback diagnostics.Finding) (contracts.TypedNormalizeResult[dnd.ItemRegistry], error) { func retryResultWithFallback(value dnd.ItemRegistry, findings, advisoryFindings []diagnostics.Finding, candidate *contracts.ModelCandidate, reasonCode, correctionGuidance, message string, fallback diagnostics.Finding) (contracts.TypedNormalizeResult[dnd.ItemRegistry], error) {
result, err := normalizationResult(value, findings, advisoryFindings, candidate) result, err := normalizationResult(value, findings, advisoryFindings, candidate)
if err != nil { if err != nil {
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{}, err return contracts.TypedNormalizeResult[dnd.ItemRegistry]{}, err
@@ -205,7 +221,7 @@ func retryResultWithFallback(value dnd.ItemRegistry, findings, advisoryFindings
if err != nil { if err != nil {
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{}, normalizerErrorf("collect fallback diagnostic: %w", err) return contracts.TypedNormalizeResult[dnd.ItemRegistry]{}, normalizerErrorf("collect fallback diagnostic: %w", err)
} }
result.Retry = &contracts.NormalizeRetry{ReasonCode: reasonCode, Message: message, FallbackDiagnostics: fallbackGroups} result.Retry = &contracts.NormalizeRetry{ReasonCode: reasonCode, Message: message, CorrectionGuidance: correctionGuidance, FallbackDiagnostics: fallbackGroups}
return result, nil return result, nil
} }

View File

@@ -122,7 +122,7 @@ func TestNormalizeAppliesSafeAliasProposal(t *testing.T) {
{Name: "Compass of the Stars", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 20, EndUnitID: 20}}}, {Name: "Compass of the Stars", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 20, EndUnitID: 20}}},
{Name: "Gold Pieces", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 30, EndUnitID: 30}}}, {Name: "Gold Pieces", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 30, EndUnitID: 30}}},
}} }}
client := &recordingNormalizerClient{response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2}]}`} client := &recordingNormalizerClient{response: `{"duplicate_groups":[{"candidate_numbers":[1,2],"canonical_candidate_number":2}]}`}
result, err := newNormalizer(t, client).Normalize(context.Background(), normalizeRequestWithSource(input, doc)) result, err := newNormalizer(t, client).Normalize(context.Background(), normalizeRequestWithSource(input, doc))
if err != nil || result.Retry != nil || len(result.Value.Items) != 2 { if err != nil || result.Retry != nil || len(result.Value.Items) != 2 {
t.Fatalf("Normalize() = %#v, %v", result, err) t.Fatalf("Normalize() = %#v, %v", result, err)
@@ -159,7 +159,7 @@ func TestNormalizeAppliesCurrencyReconciliationSafely(t *testing.T) {
{Name: "Gold Piece", SourceRefs: ref(20)}, {Name: "Gold Piece", SourceRefs: ref(20)},
{Name: "Gold Pieces", SourceRefs: ref(30)}, {Name: "Gold Pieces", SourceRefs: ref(30)},
}, },
response: `{"duplicate_groups":[{"candidate_ids":[1,2,3],"canonical_candidate_id":2}]}`, response: `{"duplicate_groups":[{"candidate_numbers":[1,2,3],"canonical_candidate_number":2}]}`,
wantNames: []string{"Gold Piece"}, wantNames: []string{"Gold Piece"},
wantRefCounts: []int{3}, wantRefCounts: []int{3},
reasonCode: ReasonCodeDuplicateItemCollapsed, reasonCode: ReasonCodeDuplicateItemCollapsed,
@@ -171,7 +171,7 @@ func TestNormalizeAppliesCurrencyReconciliationSafely(t *testing.T) {
{Name: "Gold Pieces", SourceRefs: ref(10)}, {Name: "Gold Pieces", SourceRefs: ref(10)},
{Name: "Silver Pieces", SourceRefs: ref(20)}, {Name: "Silver Pieces", SourceRefs: ref(20)},
}, },
response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":1}]}`, response: `{"duplicate_groups":[{"candidate_numbers":[1,2],"canonical_candidate_number":1}]}`,
wantNames: []string{"Gold Pieces", "Silver Pieces"}, wantNames: []string{"Gold Pieces", "Silver Pieces"},
wantRefCounts: []int{1, 1}, wantRefCounts: []int{1, 1},
wantRetry: true, wantRetry: true,
@@ -184,7 +184,7 @@ func TestNormalizeAppliesCurrencyReconciliationSafely(t *testing.T) {
{Name: "Gold Pieces", SourceRefs: ref(10)}, {Name: "Gold Pieces", SourceRefs: ref(10)},
{Name: "Longsword", SourceRefs: ref(20)}, {Name: "Longsword", SourceRefs: ref(20)},
}, },
response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":1}]}`, response: `{"duplicate_groups":[{"candidate_numbers":[1,2],"canonical_candidate_number":1}]}`,
wantNames: []string{"Gold Pieces", "Longsword"}, wantNames: []string{"Gold Pieces", "Longsword"},
wantRefCounts: []int{1, 1}, wantRefCounts: []int{1, 1},
wantRetry: true, wantRetry: true,
@@ -197,7 +197,7 @@ func TestNormalizeAppliesCurrencyReconciliationSafely(t *testing.T) {
{Name: "Star Compass", SourceRefs: ref(10)}, {Name: "Star Compass", SourceRefs: ref(10)},
{Name: "Compass of the Stars", SourceRefs: ref(20)}, {Name: "Compass of the Stars", SourceRefs: ref(20)},
}, },
response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2}]}`, response: `{"duplicate_groups":[{"candidate_numbers":[1,2],"canonical_candidate_number":2}]}`,
wantNames: []string{"Compass of the Stars"}, wantNames: []string{"Compass of the Stars"},
wantRefCounts: []int{2}, wantRefCounts: []int{2},
reasonCode: ReasonCodeDuplicateItemCollapsed, reasonCode: ReasonCodeDuplicateItemCollapsed,
@@ -209,7 +209,7 @@ func TestNormalizeAppliesCurrencyReconciliationSafely(t *testing.T) {
{Name: "Gold Pieces", SourceRefs: ref(10)}, {Name: "Gold Pieces", SourceRefs: ref(10)},
{Name: "Longsword", SourceRefs: ref(20)}, {Name: "Longsword", SourceRefs: ref(20)},
}, },
response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2}]}`, response: `{"duplicate_groups":[{"candidate_numbers":[1,2],"canonical_candidate_number":2}]}`,
wantNames: []string{"Gold Pieces", "Longsword"}, wantNames: []string{"Gold Pieces", "Longsword"},
wantRefCounts: []int{1, 1}, wantRefCounts: []int{1, 1},
wantRetry: true, wantRetry: true,
@@ -253,7 +253,7 @@ func TestNormalizePreservesCandidatesForUnsafeProposalGroups(t *testing.T) {
{Name: "Compass", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 20, EndUnitID: 20}}}, {Name: "Compass", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 20, EndUnitID: 20}}},
{Name: "Rope", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 30, EndUnitID: 30}}}, {Name: "Rope", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 30, EndUnitID: 30}}},
}} }}
client := &recordingNormalizerClient{response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":1},{"candidate_ids":[2,3],"canonical_candidate_id":3},{"candidate_ids":[1,3],"canonical_candidate_id":99}]}`} client := &recordingNormalizerClient{response: `{"duplicate_groups":[{"candidate_numbers":[1,2],"canonical_candidate_number":1},{"candidate_numbers":[2,3],"canonical_candidate_number":3},{"candidate_numbers":[1,3],"canonical_candidate_number":99}]}`}
result, err := newNormalizer(t, client).Normalize(context.Background(), normalizeRequestWithSource(input, doc)) result, err := newNormalizer(t, client).Normalize(context.Background(), normalizeRequestWithSource(input, doc))
if err != nil || result.Retry == nil || len(result.Value.Items) != 3 || !strings.Contains(result.Retry.Message, "overlapping_member") || !strings.Contains(result.Retry.Message, "canonical_unknown") || strings.Contains(result.Retry.Message, "Star Compass") || len(result.Retry.Message) > 4096 { if err != nil || result.Retry == nil || len(result.Value.Items) != 3 || !strings.Contains(result.Retry.Message, "overlapping_member") || !strings.Contains(result.Retry.Message, "canonical_unknown") || strings.Contains(result.Retry.Message, "Star Compass") || len(result.Retry.Message) > 4096 {
t.Fatalf("Normalize() = %#v, %v; want deterministic retry fallback", result, err) t.Fatalf("Normalize() = %#v, %v; want deterministic retry fallback", result, err)
@@ -272,7 +272,7 @@ func TestNormalizeAppliesIndependentGroupAndCountsAllOmissions(t *testing.T) {
{Name: "Silver Pieces", SourceRefs: ref(30)}, {Name: "Silver Pieces", SourceRefs: ref(30)},
{Name: "Rope", SourceRefs: ref(10)}, {Name: "Rope", SourceRefs: ref(10)},
}} }}
client := &recordingNormalizerClient{response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2},{"candidate_ids":[3,4],"canonical_candidate_id":3},{"candidate_ids":[5,99],"canonical_candidate_id":5}]}`} client := &recordingNormalizerClient{response: `{"duplicate_groups":[{"candidate_numbers":[1,2],"canonical_candidate_number":2},{"candidate_numbers":[3,4],"canonical_candidate_number":3},{"candidate_numbers":[5,99],"canonical_candidate_number":5}]}`}
result, err := newNormalizer(t, client).Normalize(context.Background(), normalizeRequestWithSource(input, doc)) result, err := newNormalizer(t, client).Normalize(context.Background(), normalizeRequestWithSource(input, doc))
if err != nil || result.Retry == nil { if err != nil || result.Retry == nil {
t.Fatalf("Normalize() = %#v, %v; want retry with independently accepted output", result, err) t.Fatalf("Normalize() = %#v, %v; want retry with independently accepted output", result, err)
@@ -289,7 +289,10 @@ func TestNormalizeAppliesIndependentGroupAndCountsAllOmissions(t *testing.T) {
if !hasDiagnostic(result.Diagnostics, ReasonCodeDuplicateItemCollapsed, contracts.DiagnosticDispositionObservation) || !hasDiagnostic(result.Diagnostics, ReasonCodeItemSemanticProposalInvalid, contracts.DiagnosticDispositionAdvisory) { if !hasDiagnostic(result.Diagnostics, ReasonCodeDuplicateItemCollapsed, contracts.DiagnosticDispositionObservation) || !hasDiagnostic(result.Diagnostics, ReasonCodeItemSemanticProposalInvalid, contracts.DiagnosticDispositionAdvisory) {
t.Fatalf("diagnostics = %#v, want accepted and guarded-group diagnostics", result.Diagnostics) t.Fatalf("diagnostics = %#v, want accepted and guarded-group diagnostics", result.Diagnostics)
} }
if len(result.Retry.FallbackDiagnostics) != 1 || !strings.Contains(result.Retry.FallbackDiagnostics[0].Samples[0].Message, "2 proposal group(s)") { if !strings.Contains(result.Retry.CorrectionGuidance, "aliases of the same denomination") || !strings.Contains(result.Retry.CorrectionGuidance, "do not combine currency with non-currency items") || !strings.Contains(result.Retry.CorrectionGuidance, "candidate-number range 1 through 5") || !strings.Contains(result.Retry.CorrectionGuidance, "source positions") || strings.Contains(result.Retry.CorrectionGuidance, "member_unknown") || strings.Contains(result.Retry.CorrectionGuidance, ReasonCodeItemSemanticRetryProposalInvalid) {
t.Fatalf("correction guidance = %q, want shared and item-specific prose only", result.Retry.CorrectionGuidance)
}
if len(result.Retry.FallbackDiagnostics) != 1 || !strings.Contains(result.Retry.FallbackDiagnostics[0].Samples[0].Message, "2 proposal group(s)") || !strings.Contains(result.Retry.FallbackDiagnostics[0].Samples[0].Message, "preserved separately") || !strings.Contains(result.Retry.FallbackDiagnostics[0].Samples[0].Message, "same denomination") {
t.Fatalf("retry = %#v, want one guarded and one malformed group counted", result.Retry) t.Fatalf("retry = %#v, want one guarded and one malformed group counted", result.Retry)
} }
} }
@@ -321,6 +324,9 @@ func TestNormalizeRetryFallbackErrorsAndIdempotence(t *testing.T) {
if err != nil || invalid.Retry == nil || invalid.Retry.ReasonCode != ReasonCodeItemSemanticRetryProposalInvalid { if err != nil || invalid.Retry == nil || invalid.Retry.ReasonCode != ReasonCodeItemSemanticRetryProposalInvalid {
t.Fatalf("invalid result = %#v, %v", invalid, err) t.Fatalf("invalid result = %#v, %v", invalid, err)
} }
if invalid.Retry.CorrectionGuidance != "" || !strings.Contains(invalid.Retry.FallbackDiagnostics[0].Samples[0].Message, "structured-output contract") {
t.Fatalf("invalid structured retry = %#v, want feedback-free retry with final-cause fallback", invalid.Retry)
}
_, err = newNormalizer(t, &recordingNormalizerClient{err: errors.New("provider unavailable")}).Normalize(context.Background(), normalizeRequestWithSource(input, doc)) _, err = newNormalizer(t, &recordingNormalizerClient{err: errors.New("provider unavailable")}).Normalize(context.Background(), normalizeRequestWithSource(input, doc))
if err == nil || !strings.Contains(err.Error(), "provider unavailable") { if err == nil || !strings.Contains(err.Error(), "provider unavailable") {
t.Fatalf("provider error = %v", err) t.Fatalf("provider error = %v", err)
@@ -349,11 +355,11 @@ func TestRegisterPromptAssetsPreparesItemNormalizationPrompt(t *testing.T) {
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{PromptID: PromptID, PromptVersion: PromptVersion, ProfileID: "item-normalize-test", Inputs: map[string]promptkit.ArtifactRef{"candidates": promptkit.Inline(`{"candidates":[{"candidate_id":1,"label":"Rope","source_refs":[{"start_unit_id":1,"end_unit_id":1}]}]}`), "transcript": promptkit.Inline(`{"windows":[{"units":[]}]}`)}}) prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{PromptID: PromptID, PromptVersion: PromptVersion, ProfileID: "item-normalize-test", Inputs: map[string]promptkit.ArtifactRef{"candidates": promptkit.Inline(`{"candidate_number_range":{"first":1,"last":1},"candidates":[{"candidate_number":1,"label":"Rope","source_refs":[{"start_unit_id":1,"end_unit_id":1}]}]}`), "transcript": promptkit.Inline(`{"windows":[{"units":[]}]}`)}})
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
if prepared.OutputContract.SchemaPath != "semantic_reconciliation_llm.v1.json" || !strings.Contains(prepared.Messages[1].Content, "candidate_id") || !strings.Contains(prepared.Messages[1].Content, "integer") || !strings.Contains(prepared.Messages[2].Content, "currency denominations") || !strings.Contains(prepared.Messages[2].Content, "materially different item") || prepared.Messages[2].CacheControl == nil || prepared.Messages[2].CacheControl.Type != promptkit.CacheControlEphemeral { if prepared.OutputContract.SchemaPath != "semantic_reconciliation_llm.v1.json" || !strings.Contains(prepared.Messages[1].Content, "candidate_number_range") || !strings.Contains(prepared.Messages[1].Content, "start_unit_id") || !strings.Contains(prepared.Messages[2].Content, "currency denominations") || !strings.Contains(prepared.Messages[2].Content, "materially different item") || !strings.Contains(prepared.Messages[2].Content, "canonical_candidate_number") || prepared.Messages[2].CacheControl == nil || prepared.Messages[2].CacheControl.Type != promptkit.CacheControlEphemeral {
t.Fatalf("prepared prompt = %#v", prepared) t.Fatalf("prepared prompt = %#v", prepared)
} }
if prepared.Messages[4].CacheControl == nil || prepared.Messages[4].CacheControl.Type != promptkit.CacheControlEphemeral || !strings.Contains(prepared.Messages[3].Content, `"Rope"`) || strings.Contains(prepared.Messages[3].Content, `"windows"`) || !strings.Contains(prepared.Messages[4].Content, `"windows"`) || strings.Contains(prepared.Messages[4].Content, `"Rope"`) { if prepared.Messages[4].CacheControl == nil || prepared.Messages[4].CacheControl.Type != promptkit.CacheControlEphemeral || !strings.Contains(prepared.Messages[3].Content, `"Rope"`) || strings.Contains(prepared.Messages[3].Content, `"windows"`) || !strings.Contains(prepared.Messages[4].Content, `"windows"`) || strings.Contains(prepared.Messages[4].Content, `"Rope"`) {

View File

@@ -24,7 +24,7 @@ const (
Key = "dnd/location-registry" Key = "dnd/location-registry"
PromptID = "dnd.location_registry.normalize" PromptID = "dnd.location_registry.normalize"
PromptVersion = "v1" PromptVersion = "v1"
normalizationPolicy = "dnd.location_registry.normalize.v3" normalizationPolicy = "dnd.location_registry.normalize.v4"
NormalizationPolicy = normalizationPolicy NormalizationPolicy = normalizationPolicy
ReasonCodeLocationFieldsNormalized = "location_fields_normalized" ReasonCodeLocationFieldsNormalized = "location_fields_normalized"
@@ -129,7 +129,7 @@ func (n *Normalizer) Normalize(ctx context.Context, req contracts.TypedNormalize
case semanticreconcile.SkippedInsufficientCandidates: case semanticreconcile.SkippedInsufficientCandidates:
return normalizationResult(deterministic, findings, nil) return normalizationResult(deterministic, findings, nil)
case semanticreconcile.SkippedLimitExceeded: case semanticreconcile.SkippedLimitExceeded:
return fallbackResult(deterministic, findings, semanticFallbackFinding(-1)) return fallbackResult(deterministic, findings, semanticFallbackFinding(-1, "the candidate set exceeded the configured semantic-reconciliation limits"))
case semanticreconcile.RetryableInvalidStructuredOutput: case semanticreconcile.RetryableInvalidStructuredOutput:
return n.invalidStructuredResult(deterministic, findings) return n.invalidStructuredResult(deterministic, findings)
case semanticreconcile.Complete, semanticreconcile.RetryableDiscardedProposalGroups: case semanticreconcile.Complete, semanticreconcile.RetryableDiscardedProposalGroups:
@@ -149,17 +149,29 @@ func (n *Normalizer) Normalize(ctx context.Context, req contracts.TypedNormalize
} }
func (n *Normalizer) invalidStructuredResult(value dnd.LocationRegistry, findings []diagnostics.Finding) (contracts.TypedNormalizeResult[dnd.LocationRegistry], error) { func (n *Normalizer) invalidStructuredResult(value dnd.LocationRegistry, findings []diagnostics.Finding) (contracts.TypedNormalizeResult[dnd.LocationRegistry], error) {
return retryResultWithFallback(value, findings, nil, "semantic proposal requires retry: invalid structured output", semanticFallbackFinding(-1)) return retryResultWithFallback(value, findings, nil, "", "semantic proposal requires retry: invalid structured output", semanticFallbackFinding(-1, "the final model response did not satisfy the structured-output contract"))
} }
func retryResult(value dnd.LocationRegistry, findings []diagnostics.Finding, reconciliation semanticreconcile.Result) (contracts.TypedNormalizeResult[dnd.LocationRegistry], error) { func retryResult(value dnd.LocationRegistry, findings []diagnostics.Finding, reconciliation semanticreconcile.Result) (contracts.TypedNormalizeResult[dnd.LocationRegistry], error) {
return retryResultWithFallback(value, findings, reconciliation.ModelCandidate(), diagnostics.Aggregate("semantic proposal requires retry", semanticreconcile.IssueDetails(reconciliation.Issues())), semanticFallbackFinding(reconciliation.DiscardedGroupCount())) issues := reconciliation.Issues()
correctionDetails, err := reconciliation.CorrectionDetails()
if err != nil {
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{}, normalizerErrorf("build semantic correction details: %w", err)
}
correctionGuidance, err := reconciliation.CorrectionGuidance()
if err != nil {
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{}, normalizerErrorf("build semantic correction guidance: %w", err)
}
return retryResultWithFallback(value, findings, reconciliation.ModelCandidate(), correctionGuidance, diagnostics.Aggregate("semantic proposal requires retry", semanticreconcile.IssueDetails(issues)), semanticFallbackFinding(reconciliation.DiscardedGroupCount(), correctionDetails...))
} }
func semanticFallbackFinding(discarded int) diagnostics.Finding { func semanticFallbackFinding(discarded int, details ...string) diagnostics.Finding {
message := "semantic proposal could not be applied" message := "semantic reconciliation could not be completed; all affected candidates were preserved separately"
if discarded >= 0 { if discarded >= 0 {
message = fmt.Sprintf("%d proposal group(s) omitted after semantic proposal retry exhaustion", discarded) message = fmt.Sprintf("%d proposal group(s) omitted from the final semantic proposal; all affected candidates were preserved separately", discarded)
}
if len(details) > 0 {
message = diagnostics.Aggregate(message, details)
} }
return diagnostics.Finding{Scope: "locations", ReasonCode: ReasonCodeLocationSemanticReconciliationExhausted, Message: message} return diagnostics.Finding{Scope: "locations", ReasonCode: ReasonCodeLocationSemanticReconciliationExhausted, Message: message}
} }
@@ -185,7 +197,7 @@ func fallbackResult(value dnd.LocationRegistry, findings []diagnostics.Finding,
return result, nil return result, nil
} }
func retryResultWithFallback(value dnd.LocationRegistry, findings []diagnostics.Finding, candidate *contracts.ModelCandidate, message string, fallback diagnostics.Finding) (contracts.TypedNormalizeResult[dnd.LocationRegistry], error) { func retryResultWithFallback(value dnd.LocationRegistry, findings []diagnostics.Finding, candidate *contracts.ModelCandidate, correctionGuidance, message string, fallback diagnostics.Finding) (contracts.TypedNormalizeResult[dnd.LocationRegistry], error) {
result, err := normalizationResult(value, findings, candidate) result, err := normalizationResult(value, findings, candidate)
if err != nil { if err != nil {
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{}, err return contracts.TypedNormalizeResult[dnd.LocationRegistry]{}, err
@@ -194,7 +206,7 @@ func retryResultWithFallback(value dnd.LocationRegistry, findings []diagnostics.
if err != nil { if err != nil {
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{}, normalizerErrorf("collect fallback diagnostic: %w", err) return contracts.TypedNormalizeResult[dnd.LocationRegistry]{}, normalizerErrorf("collect fallback diagnostic: %w", err)
} }
result.Retry = &contracts.NormalizeRetry{ReasonCode: ReasonCodeLocationSemanticProposalInvalid, Message: message, FallbackDiagnostics: fallbackGroups} result.Retry = &contracts.NormalizeRetry{ReasonCode: ReasonCodeLocationSemanticProposalInvalid, Message: message, CorrectionGuidance: correctionGuidance, FallbackDiagnostics: fallbackGroups}
return result, nil return result, nil
} }

View File

@@ -95,7 +95,7 @@ func BenchmarkExactDuplicateGroupsManyDistinct(b *testing.B) {
} }
func TestNormalizeAppliesSafeAliasGroupAndUsesContextualInputs(t *testing.T) { func TestNormalizeAppliesSafeAliasGroupAndUsesContextualInputs(t *testing.T) {
client := &recordingLocationNormalizerClient{response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2}]}`} client := &recordingLocationNormalizerClient{response: `{"duplicate_groups":[{"candidate_numbers":[1,2],"canonical_candidate_number":2}]}`}
doc := semanticDocument() doc := semanticDocument()
input := dnd.LocationRegistry{Locations: []dnd.Location{ input := dnd.LocationRegistry{Locations: []dnd.Location{
{Name: "Old Mill", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 10, EndUnitID: 10}}}, {Name: "Old Mill", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 10, EndUnitID: 10}}},
@@ -123,12 +123,15 @@ func TestNormalizeRejectsUnsafeAndOverlappingGroupsWithoutLosingCandidates(t *te
{Name: "Mill", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 20, EndUnitID: 20}}}, {Name: "Mill", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 20, EndUnitID: 20}}},
{Name: "Tavern", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 30, EndUnitID: 30}}}, {Name: "Tavern", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 30, EndUnitID: 30}}},
}} }}
client := &recordingLocationNormalizerClient{response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":1},{"candidate_ids":[2,3],"canonical_candidate_id":3}]}`} client := &recordingLocationNormalizerClient{response: `{"duplicate_groups":[{"candidate_numbers":[1,2],"canonical_candidate_number":1},{"candidate_numbers":[2,3],"canonical_candidate_number":3}]}`}
result, err := newNormalizer(t, client).Normalize(context.Background(), normalizeRequestWithSource(input, doc)) result, err := newNormalizer(t, client).Normalize(context.Background(), normalizeRequestWithSource(input, doc))
if err != nil || result.Retry == nil || len(result.Value.Locations) != 3 || !strings.Contains(result.Retry.Message, "overlapping_member") { if err != nil || result.Retry == nil || len(result.Value.Locations) != 3 || !strings.Contains(result.Retry.Message, "overlapping_member") {
t.Fatalf("Normalize() = %#v, %v; want safe retry fallback", result, err) t.Fatalf("Normalize() = %#v, %v; want safe retry fallback", result, err)
} }
if len(result.Retry.FallbackDiagnostics) != 1 || !strings.Contains(result.Retry.FallbackDiagnostics[0].Samples[0].Message, "2 proposal group") { if !strings.Contains(result.Retry.CorrectionGuidance, "Duplicate groups 1, 2") || !strings.Contains(result.Retry.CorrectionGuidance, "at most one duplicate group") || strings.Contains(result.Retry.CorrectionGuidance, "overlapping_member") {
t.Fatalf("correction guidance = %q, want model-facing overlap explanation", result.Retry.CorrectionGuidance)
}
if len(result.Retry.FallbackDiagnostics) != 1 || !strings.Contains(result.Retry.FallbackDiagnostics[0].Samples[0].Message, "2 proposal group") || !strings.Contains(result.Retry.FallbackDiagnostics[0].Samples[0].Message, "preserved separately") {
t.Fatalf("fallback diagnostics = %#v", result.Retry.FallbackDiagnostics) t.Fatalf("fallback diagnostics = %#v", result.Retry.FallbackDiagnostics)
} }
} }
@@ -149,14 +152,14 @@ func TestReconciliationCandidatesKeepSameNameEvidenceDistinct(t *testing.T) {
} }
var candidateInput struct { var candidateInput struct {
Candidates []struct { Candidates []struct {
CandidateID int `json:"candidate_id"` CandidateNumber int `json:"candidate_number"`
Label string `json:"label"` Label string `json:"label"`
} `json:"candidates"` } `json:"candidates"`
} }
if err := json.Unmarshal(preparation.Materials()["candidates"].Content, &candidateInput); err != nil { if err := json.Unmarshal(preparation.Materials()["candidates"].Content, &candidateInput); err != nil {
t.Fatal(err) t.Fatal(err)
} }
if len(candidateInput.Candidates) != 2 || candidateInput.Candidates[0].CandidateID != 1 || candidateInput.Candidates[1].CandidateID != 2 || candidateInput.Candidates[0].Label != "The Tavern" || candidateInput.Candidates[1].Label != "The Tavern" { if len(candidateInput.Candidates) != 2 || candidateInput.Candidates[0].CandidateNumber != 1 || candidateInput.Candidates[1].CandidateNumber != 2 || candidateInput.Candidates[0].Label != "The Tavern" || candidateInput.Candidates[1].Label != "The Tavern" {
t.Fatalf("candidate input = %#v, want distinct integer handles for equal names", candidateInput) t.Fatalf("candidate input = %#v, want distinct integer handles for equal names", candidateInput)
} }
} }
@@ -188,6 +191,9 @@ func TestNormalizeHandlesRetryFallbackAndErrors(t *testing.T) {
if err != nil || invalid.Retry == nil || invalid.Retry.ReasonCode != ReasonCodeLocationSemanticProposalInvalid { if err != nil || invalid.Retry == nil || invalid.Retry.ReasonCode != ReasonCodeLocationSemanticProposalInvalid {
t.Fatalf("invalid result = %#v, %v", invalid, err) t.Fatalf("invalid result = %#v, %v", invalid, err)
} }
if invalid.Retry.CorrectionGuidance != "" || !strings.Contains(invalid.Retry.FallbackDiagnostics[0].Samples[0].Message, "structured-output contract") || !strings.Contains(invalid.Retry.FallbackDiagnostics[0].Samples[0].Message, "preserved separately") {
t.Fatalf("invalid structured retry = %#v, want feedback-free retry and informative fallback", invalid.Retry)
}
_, err = newNormalizer(t, &recordingLocationNormalizerClient{err: errors.New("provider unavailable")}).Normalize(context.Background(), normalizeRequestWithSource(input, doc)) _, err = newNormalizer(t, &recordingLocationNormalizerClient{err: errors.New("provider unavailable")}).Normalize(context.Background(), normalizeRequestWithSource(input, doc))
if err == nil || !strings.Contains(err.Error(), "provider unavailable") { if err == nil || !strings.Contains(err.Error(), "provider unavailable") {
t.Fatalf("provider error = %v", err) t.Fatalf("provider error = %v", err)

View File

@@ -28,11 +28,11 @@ func TestRegisterPromptAssetsPreparesLocationNormalizationPrompt(t *testing.T) {
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{PromptID: PromptID, PromptVersion: PromptVersion, ProfileID: "location-normalize-test", Inputs: map[string]promptkit.ArtifactRef{"candidates": promptkit.Inline(`{"candidates":[{"candidate_id":1,"label":"The Tavern","source_refs":[{"start_unit_id":1,"end_unit_id":1}]}]}`), "transcript": promptkit.Inline(`{"windows":[{"units":[]}]}`)}}) prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{PromptID: PromptID, PromptVersion: PromptVersion, ProfileID: "location-normalize-test", Inputs: map[string]promptkit.ArtifactRef{"candidates": promptkit.Inline(`{"candidate_number_range":{"first":1,"last":1},"candidates":[{"candidate_number":1,"label":"The Tavern","source_refs":[{"start_unit_id":1,"end_unit_id":1}]}]}`), "transcript": promptkit.Inline(`{"windows":[{"units":[]}]}`)}})
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
if prepared.OutputContract.SchemaPath != "semantic_reconciliation_llm.v1.json" || !strings.Contains(prepared.Messages[1].Content, "candidate_id") || !strings.Contains(prepared.Messages[1].Content, "integer") || !strings.Contains(prepared.Messages[2].Content, "same physical place") || !strings.Contains(prepared.Messages[2].Content, "parent and child places") { if prepared.OutputContract.SchemaPath != "semantic_reconciliation_llm.v1.json" || !strings.Contains(prepared.Messages[1].Content, "candidate_number_range") || !strings.Contains(prepared.Messages[1].Content, "start_unit_id") || !strings.Contains(prepared.Messages[2].Content, "same physical place") || !strings.Contains(prepared.Messages[2].Content, "parent and child places") || !strings.Contains(prepared.Messages[2].Content, "canonical_candidate_number") {
t.Fatalf("prepared prompt = %#v", prepared) t.Fatalf("prepared prompt = %#v", prepared)
} }
for _, index := range []int{2, 4} { for _, index := range []int{2, 4} {

View File

@@ -23,7 +23,7 @@ const (
Key = "dnd/npc-registry" Key = "dnd/npc-registry"
PromptID = "dnd.npc_registry.normalize" PromptID = "dnd.npc_registry.normalize"
PromptVersion = "v1" PromptVersion = "v1"
normalizationPolicy = "dnd.npc_registry.normalize.v5" normalizationPolicy = "dnd.npc_registry.normalize.v6"
NormalizationPolicy = normalizationPolicy NormalizationPolicy = normalizationPolicy
ReasonCodeNPCFieldsNormalized = "npc_fields_normalized" ReasonCodeNPCFieldsNormalized = "npc_fields_normalized"
@@ -128,7 +128,7 @@ func (n *Normalizer) Normalize(ctx context.Context, req contracts.TypedNormalize
case semanticreconcile.SkippedInsufficientCandidates: case semanticreconcile.SkippedInsufficientCandidates:
return normalizationResult(deterministic, findings, nil) return normalizationResult(deterministic, findings, nil)
case semanticreconcile.SkippedLimitExceeded: case semanticreconcile.SkippedLimitExceeded:
return fallbackResult(deterministic, findings, semanticFallbackFinding(-1)) return fallbackResult(deterministic, findings, semanticFallbackFinding(-1, "the candidate set exceeded the configured semantic-reconciliation limits"))
case semanticreconcile.RetryableInvalidStructuredOutput: case semanticreconcile.RetryableInvalidStructuredOutput:
return n.invalidStructuredResult(deterministic, findings) return n.invalidStructuredResult(deterministic, findings)
case semanticreconcile.Complete, semanticreconcile.RetryableDiscardedProposalGroups: case semanticreconcile.Complete, semanticreconcile.RetryableDiscardedProposalGroups:
@@ -148,17 +148,29 @@ func (n *Normalizer) Normalize(ctx context.Context, req contracts.TypedNormalize
} }
func (n *Normalizer) invalidStructuredResult(value dnd.NPCRegistry, findings []diagnostics.Finding) (contracts.TypedNormalizeResult[dnd.NPCRegistry], error) { func (n *Normalizer) invalidStructuredResult(value dnd.NPCRegistry, findings []diagnostics.Finding) (contracts.TypedNormalizeResult[dnd.NPCRegistry], error) {
return retryResultWithFallback(value, findings, nil, "semantic proposal requires retry: invalid structured output", semanticFallbackFinding(-1)) return retryResultWithFallback(value, findings, nil, "", "semantic proposal requires retry: invalid structured output", semanticFallbackFinding(-1, "the final model response did not satisfy the structured-output contract"))
} }
func retryResult(value dnd.NPCRegistry, findings []diagnostics.Finding, reconciliation semanticreconcile.Result) (contracts.TypedNormalizeResult[dnd.NPCRegistry], error) { func retryResult(value dnd.NPCRegistry, findings []diagnostics.Finding, reconciliation semanticreconcile.Result) (contracts.TypedNormalizeResult[dnd.NPCRegistry], error) {
return retryResultWithFallback(value, findings, reconciliation.ModelCandidate(), diagnostics.Aggregate("semantic proposal requires retry", semanticreconcile.IssueDetails(reconciliation.Issues())), semanticFallbackFinding(reconciliation.DiscardedGroupCount())) issues := reconciliation.Issues()
correctionDetails, err := reconciliation.CorrectionDetails()
if err != nil {
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{}, normalizerErrorf("build semantic correction details: %w", err)
}
correctionGuidance, err := reconciliation.CorrectionGuidance()
if err != nil {
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{}, normalizerErrorf("build semantic correction guidance: %w", err)
}
return retryResultWithFallback(value, findings, reconciliation.ModelCandidate(), correctionGuidance, diagnostics.Aggregate("semantic proposal requires retry", semanticreconcile.IssueDetails(issues)), semanticFallbackFinding(reconciliation.DiscardedGroupCount(), correctionDetails...))
} }
func semanticFallbackFinding(discardedGroups int) diagnostics.Finding { func semanticFallbackFinding(discardedGroups int, details ...string) diagnostics.Finding {
message := "semantic proposal could not be applied" message := "semantic reconciliation could not be completed; all affected candidates were preserved separately"
if discardedGroups >= 0 { if discardedGroups >= 0 {
message = fmt.Sprintf("%d proposal group(s) omitted after semantic proposal retry exhaustion", discardedGroups) message = fmt.Sprintf("%d proposal group(s) omitted from the final semantic proposal; all affected candidates were preserved separately", discardedGroups)
}
if len(details) > 0 {
message = diagnostics.Aggregate(message, details)
} }
return diagnostics.Finding{Scope: "npcs", ReasonCode: ReasonCodeNPCSemanticReconciliationExhausted, Message: message} return diagnostics.Finding{Scope: "npcs", ReasonCode: ReasonCodeNPCSemanticReconciliationExhausted, Message: message}
} }
@@ -184,7 +196,7 @@ func fallbackResult(value dnd.NPCRegistry, findings []diagnostics.Finding, fallb
return result, nil return result, nil
} }
func retryResultWithFallback(value dnd.NPCRegistry, findings []diagnostics.Finding, candidate *contracts.ModelCandidate, message string, fallback diagnostics.Finding) (contracts.TypedNormalizeResult[dnd.NPCRegistry], error) { func retryResultWithFallback(value dnd.NPCRegistry, findings []diagnostics.Finding, candidate *contracts.ModelCandidate, correctionGuidance, message string, fallback diagnostics.Finding) (contracts.TypedNormalizeResult[dnd.NPCRegistry], error) {
result, err := normalizationResult(value, findings, candidate) result, err := normalizationResult(value, findings, candidate)
if err != nil { if err != nil {
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{}, err return contracts.TypedNormalizeResult[dnd.NPCRegistry]{}, err
@@ -193,7 +205,7 @@ func retryResultWithFallback(value dnd.NPCRegistry, findings []diagnostics.Findi
if err != nil { if err != nil {
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{}, normalizerErrorf("collect fallback diagnostic: %w", err) return contracts.TypedNormalizeResult[dnd.NPCRegistry]{}, normalizerErrorf("collect fallback diagnostic: %w", err)
} }
result.Retry = &contracts.NormalizeRetry{ReasonCode: ReasonCodeNPCSemanticProposalInvalid, Message: message, FallbackDiagnostics: fallbackGroups} result.Retry = &contracts.NormalizeRetry{ReasonCode: ReasonCodeNPCSemanticProposalInvalid, Message: message, CorrectionGuidance: correctionGuidance, FallbackDiagnostics: fallbackGroups}
return result, nil return result, nil
} }

View File

@@ -36,7 +36,7 @@ func TestRegisterPromptAssetsPreparesNormalizationPrompt(t *testing.T) {
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{ prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{
PromptID: PromptID, PromptVersion: PromptVersion, ProfileID: "normalize-test-profile", PromptID: PromptID, PromptVersion: PromptVersion, ProfileID: "normalize-test-profile",
Inputs: map[string]promptkit.ArtifactRef{ Inputs: map[string]promptkit.ArtifactRef{
"candidates": promptkit.Inline(`{"candidates":[{"candidate_id":1,"label":"Mira","source_refs":[{"start_unit_id":1,"end_unit_id":1}]}]}`), "candidates": promptkit.Inline(`{"candidate_number_range":{"first":1,"last":1},"candidates":[{"candidate_number":1,"label":"Mira","source_refs":[{"start_unit_id":1,"end_unit_id":1}]}]}`),
"transcript": promptkit.Inline(`{"windows":[{"units":[]}]}`), "transcript": promptkit.Inline(`{"windows":[{"units":[]}]}`),
}, },
}) })
@@ -49,10 +49,10 @@ func TestRegisterPromptAssetsPreparesNormalizationPrompt(t *testing.T) {
if prepared.Messages[0].Role != "system" { if prepared.Messages[0].Role != "system" {
t.Fatalf("initial message role = %q, want system", prepared.Messages[0].Role) t.Fatalf("initial message role = %q, want system", prepared.Messages[0].Role)
} }
if !strings.Contains(prepared.Messages[1].Content, "candidate_id") || !strings.Contains(prepared.Messages[1].Content, "integer") { if !strings.Contains(prepared.Messages[1].Content, "candidate_number") || !strings.Contains(prepared.Messages[1].Content, "integer") {
t.Fatalf("protocol message = %q, want shared integer-handle protocol", prepared.Messages[1].Content) t.Fatalf("protocol message = %q, want shared integer-handle protocol", prepared.Messages[1].Content)
} }
if !strings.Contains(prepared.Messages[2].Content, "same individual") || strings.Contains(prepared.Messages[2].Content, "source ranges") { if !strings.Contains(prepared.Messages[2].Content, "same individual") || !strings.Contains(prepared.Messages[2].Content, "canonical_candidate_number") || strings.Contains(prepared.Messages[2].Content, "source ranges") {
t.Fatalf("NPC policy message = %q, want domain distinctions without copied ranges", prepared.Messages[2].Content) t.Fatalf("NPC policy message = %q, want domain distinctions without copied ranges", prepared.Messages[2].Content)
} }
for _, index := range []int{2, 4} { for _, index := range []int{2, 4} {

View File

@@ -35,7 +35,7 @@ func TestNormalizeSkipsSemanticCompletionWithoutTwoEligibleCandidates(t *testing
} }
func TestNormalizeAppliesSafeProposalAndUsesPrivateInputs(t *testing.T) { func TestNormalizeAppliesSafeProposalAndUsesPrivateInputs(t *testing.T) {
client := &recordingNPCNormalizerClient{response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2}]}`} client := &recordingNPCNormalizerClient{response: `{"duplicate_groups":[{"candidate_numbers":[1,2],"canonical_candidate_number":2}]}`}
normalizer := newNormalizer(t, client) normalizer := newNormalizer(t, client)
doc := semanticDocument() doc := semanticDocument()
input := dnd.NPCRegistry{NPCs: []dnd.NPC{ input := dnd.NPCRegistry{NPCs: []dnd.NPC{
@@ -88,19 +88,19 @@ func TestNormalizeAppliesSafeProposalAndUsesPrivateInputs(t *testing.T) {
} }
var visible struct { var visible struct {
Candidates []struct { Candidates []struct {
CandidateID int `json:"candidate_id"` CandidateNumber int `json:"candidate_number"`
} `json:"candidates"` } `json:"candidates"`
} }
if err := json.Unmarshal(completion.Inputs["candidates"].Content, &visible); err != nil { if err := json.Unmarshal(completion.Inputs["candidates"].Content, &visible); err != nil {
t.Fatalf("decode candidates: %v", err) t.Fatalf("decode candidates: %v", err)
} }
if got := []int{visible.Candidates[0].CandidateID, visible.Candidates[1].CandidateID, visible.Candidates[2].CandidateID}; !reflect.DeepEqual(got, []int{1, 2, 3}) { if got := []int{visible.Candidates[0].CandidateNumber, visible.Candidates[1].CandidateNumber, visible.Candidates[2].CandidateNumber}; !reflect.DeepEqual(got, []int{1, 2, 3}) {
t.Fatalf("candidate IDs = %v, want contiguous request-local handles", got) t.Fatalf("candidate numbers = %v, want contiguous request-local handles", got)
} }
} }
func TestNormalizeUnsafeProposalReturnsSafeRetryFallback(t *testing.T) { func TestNormalizeUnsafeProposalReturnsSafeRetryFallback(t *testing.T) {
client := &recordingNPCNormalizerClient{response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2},{"candidate_ids":[3,99],"canonical_candidate_id":3}]}`} client := &recordingNPCNormalizerClient{response: `{"duplicate_groups":[{"candidate_numbers":[1,2],"canonical_candidate_number":2},{"candidate_numbers":[3,99],"canonical_candidate_number":3}]}`}
normalizer := newNormalizer(t, client) normalizer := newNormalizer(t, client)
doc := semanticDocument() doc := semanticDocument()
input := dnd.NPCRegistry{NPCs: []dnd.NPC{ input := dnd.NPCRegistry{NPCs: []dnd.NPC{
@@ -115,10 +115,13 @@ func TestNormalizeUnsafeProposalReturnsSafeRetryFallback(t *testing.T) {
if result.Retry.ReasonCode != ReasonCodeNPCSemanticProposalInvalid || !strings.Contains(result.Retry.Message, "group 1: member_unknown") { if result.Retry.ReasonCode != ReasonCodeNPCSemanticProposalInvalid || !strings.Contains(result.Retry.Message, "group 1: member_unknown") {
t.Fatalf("retry = %#v, want bounded invalid-proposal diagnostics", result.Retry) t.Fatalf("retry = %#v, want bounded invalid-proposal diagnostics", result.Retry)
} }
if !strings.Contains(result.Retry.CorrectionGuidance, "Duplicate group 2") || !strings.Contains(result.Retry.CorrectionGuidance, "candidate-number range 1 through 3") || !strings.Contains(result.Retry.CorrectionGuidance, "source positions") || strings.Contains(result.Retry.CorrectionGuidance, "member_unknown") || strings.Contains(result.Retry.CorrectionGuidance, ReasonCodeNPCSemanticProposalInvalid) {
t.Fatalf("correction guidance = %q, want contextual prose without internal labels", result.Retry.CorrectionGuidance)
}
if len(result.Value.NPCs) != 2 || result.Value.NPCs[0].Name != "Mira Thorn" || result.Value.NPCs[1].Name != "Captain Vale" { if len(result.Value.NPCs) != 2 || result.Value.NPCs[0].Name != "Mira Thorn" || result.Value.NPCs[1].Name != "Captain Vale" {
t.Fatalf("fallback NPCs = %#v, want independently safe group applied", result.Value.NPCs) t.Fatalf("fallback NPCs = %#v, want independently safe group applied", result.Value.NPCs)
} }
if len(result.Retry.FallbackDiagnostics) != 1 || result.Retry.FallbackDiagnostics[0].ReasonCode != ReasonCodeNPCSemanticReconciliationExhausted || !strings.Contains(result.Retry.FallbackDiagnostics[0].Samples[0].Message, "1 proposal group") { if len(result.Retry.FallbackDiagnostics) != 1 || result.Retry.FallbackDiagnostics[0].ReasonCode != ReasonCodeNPCSemanticReconciliationExhausted || !strings.Contains(result.Retry.FallbackDiagnostics[0].Samples[0].Message, "1 proposal group") || !strings.Contains(result.Retry.FallbackDiagnostics[0].Samples[0].Message, "preserved separately") || !strings.Contains(result.Retry.FallbackDiagnostics[0].Samples[0].Message, "candidate-number range 1 through 3") {
t.Fatalf("fallback diagnostics = %#v, want exact omitted-group warning", result.Retry.FallbackDiagnostics) t.Fatalf("fallback diagnostics = %#v, want exact omitted-group warning", result.Retry.FallbackDiagnostics)
} }
} }
@@ -134,6 +137,9 @@ func TestNormalizeInvalidStructuredOutputAndOperationalErrorsRemainDistinct(t *t
if err != nil || result.Retry == nil || result.Retry.ReasonCode != ReasonCodeNPCSemanticProposalInvalid || len(result.Value.NPCs) != 2 { if err != nil || result.Retry == nil || result.Retry.ReasonCode != ReasonCodeNPCSemanticProposalInvalid || len(result.Value.NPCs) != 2 {
t.Fatalf("invalid structured result = %#v, %v; want deterministic retry fallback", result, err) t.Fatalf("invalid structured result = %#v, %v; want deterministic retry fallback", result, err)
} }
if result.Retry.CorrectionGuidance != "" || !strings.Contains(result.Retry.FallbackDiagnostics[0].Samples[0].Message, "structured-output contract") || !strings.Contains(result.Retry.FallbackDiagnostics[0].Samples[0].Message, "preserved separately") {
t.Fatalf("invalid structured retry = %#v, want feedback-free retry and informative fallback", result.Retry)
}
operational := errors.New("provider unavailable") operational := errors.New("provider unavailable")
if _, err := newNormalizer(t, &recordingNPCNormalizerClient{err: operational}).Normalize(context.Background(), normalizeRequestWithSource(input, doc)); !errors.Is(err, operational) || errors.Is(err, contracts.ErrInvalidStructuredOutput) { if _, err := newNormalizer(t, &recordingNPCNormalizerClient{err: operational}).Normalize(context.Background(), normalizeRequestWithSource(input, doc)); !errors.Is(err, operational) || errors.Is(err, contracts.ErrInvalidStructuredOutput) {
t.Fatalf("operational completion error = %v, want ordinary error", err) t.Fatalf("operational completion error = %v, want ordinary error", err)
@@ -178,8 +184,8 @@ func TestNormalizeRedactsContextMaterialFailures(t *testing.T) {
func TestNormalizeDoesNotAccumulateSafeGroupsAcrossAttempts(t *testing.T) { func TestNormalizeDoesNotAccumulateSafeGroupsAcrossAttempts(t *testing.T) {
client := &recordingNPCNormalizerClient{responses: []string{ client := &recordingNPCNormalizerClient{responses: []string{
`{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2},{"candidate_ids":[3,99],"canonical_candidate_id":3}]}`, `{"duplicate_groups":[{"candidate_numbers":[1,2],"canonical_candidate_number":2},{"candidate_numbers":[3,99],"canonical_candidate_number":3}]}`,
`{"duplicate_groups":[{"candidate_ids":[1,3],"canonical_candidate_id":3}]}`, `{"duplicate_groups":[{"candidate_numbers":[1,3],"canonical_candidate_number":3}]}`,
}} }}
normalizer := newNormalizer(t, client) normalizer := newNormalizer(t, client)
doc := semanticDocument() doc := semanticDocument()
@@ -202,7 +208,7 @@ func TestNormalizeDoesNotAccumulateSafeGroupsAcrossAttempts(t *testing.T) {
} }
func TestNormalizeRetainsRecordsWhenAProposalUsesAnUnknownHandle(t *testing.T) { func TestNormalizeRetainsRecordsWhenAProposalUsesAnUnknownHandle(t *testing.T) {
client := &recordingNPCNormalizerClient{response: `{"duplicate_groups":[{"candidate_ids":[1,99],"canonical_candidate_id":1}]}`} client := &recordingNPCNormalizerClient{response: `{"duplicate_groups":[{"candidate_numbers":[1,99],"canonical_candidate_number":1}]}`}
normalizer := newNormalizer(t, client) normalizer := newNormalizer(t, client)
doc := semanticDocument() doc := semanticDocument()
input := dnd.NPCRegistry{NPCs: []dnd.NPC{ input := dnd.NPCRegistry{NPCs: []dnd.NPC{
@@ -235,14 +241,14 @@ func TestReconciliationCandidatesKeepEqualContextualDescriptorsDistinct(t *testi
} }
var candidateInput struct { var candidateInput struct {
Candidates []struct { Candidates []struct {
CandidateID int `json:"candidate_id"` CandidateNumber int `json:"candidate_number"`
Label string `json:"label"` Label string `json:"label"`
} `json:"candidates"` } `json:"candidates"`
} }
if err := json.Unmarshal(preparation.Materials()["candidates"].Content, &candidateInput); err != nil { if err := json.Unmarshal(preparation.Materials()["candidates"].Content, &candidateInput); err != nil {
t.Fatal(err) t.Fatal(err)
} }
if len(candidateInput.Candidates) != 2 || candidateInput.Candidates[0].CandidateID != 1 || candidateInput.Candidates[1].CandidateID != 2 || candidateInput.Candidates[0].Label != "The Guard" || candidateInput.Candidates[1].Label != "The Guard" { if len(candidateInput.Candidates) != 2 || candidateInput.Candidates[0].CandidateNumber != 1 || candidateInput.Candidates[1].CandidateNumber != 2 || candidateInput.Candidates[0].Label != "The Guard" || candidateInput.Candidates[1].Label != "The Guard" {
t.Fatalf("candidate input = %#v, want distinct integer handles for equal descriptors", candidateInput) t.Fatalf("candidate input = %#v, want distinct integer handles for equal descriptors", candidateInput)
} }
} }

View File

@@ -35,6 +35,7 @@ import (
npcnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/npcregistry" npcnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/npcregistry"
scenedescriptionnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/scenedescriptions" scenedescriptionnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/scenedescriptions"
spellnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/spells" spellnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/spells"
combatsemantics "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/validate/scenedescriptions/combat_semantics"
"gitea.maximumdirect.net/eric/notarius/internal/modules/generic/merge/appendorder" "gitea.maximumdirect.net/eric/notarius/internal/modules/generic/merge/appendorder"
"gitea.maximumdirect.net/eric/notarius/internal/modules/generic/normalize/noop" "gitea.maximumdirect.net/eric/notarius/internal/modules/generic/normalize/noop"
) )
@@ -157,6 +158,7 @@ func registerPromptAssets(assets *llm.AssetRegistry) error {
{name: "item registry normalization prompt assets", register: func() error { return itemregistrynormalize.RegisterPromptAssets(assets) }}, {name: "item registry normalization prompt assets", register: func() error { return itemregistrynormalize.RegisterPromptAssets(assets) }},
{name: "npc occurrences prompt assets", register: func() error { return occurrenceextract.RegisterPromptAssets(assets) }}, {name: "npc occurrences prompt assets", register: func() error { return occurrenceextract.RegisterPromptAssets(assets) }},
{name: "scene descriptions prompt assets", register: func() error { return scenedescriptionextract.RegisterPromptAssets(assets) }}, {name: "scene descriptions prompt assets", register: func() error { return scenedescriptionextract.RegisterPromptAssets(assets) }},
{name: "scene combat-semantics validator prompt assets", register: func() error { return combatsemantics.RegisterPromptAssets(assets) }},
{name: "location registry prompt assets", register: func() error { return locationextract.RegisterPromptAssets(assets) }}, {name: "location registry prompt assets", register: func() error { return locationextract.RegisterPromptAssets(assets) }},
{name: "location normalization prompt assets", register: func() error { return locationnormalize.RegisterPromptAssets(assets) }}, {name: "location normalization prompt assets", register: func() error { return locationnormalize.RegisterPromptAssets(assets) }},
{name: "location occurrences prompt assets", register: func() error { return locationoccurrenceextract.RegisterPromptAssets(assets) }}, {name: "location occurrences prompt assets", register: func() error { return locationoccurrenceextract.RegisterPromptAssets(assets) }},

View File

@@ -30,6 +30,8 @@ func TestExtractionPromptComposition(t *testing.T) {
npcSentinel = "npc-registry-sentinel" npcSentinel = "npc-registry-sentinel"
catalogSentinel = "spell-catalog-sentinel" catalogSentinel = "spell-catalog-sentinel"
evidenceSentinel = "Transcript units are the only evidence" evidenceSentinel = "Transcript units are the only evidence"
endpointOrderSentinel = "according to transcript order"
combatPolicySentinel = "substantive active combat materially organizes the"
) )
registry := llm.NewAssetRegistry() registry := llm.NewAssetRegistry()
if err := registerPromptAssets(registry); err != nil { if err := registerPromptAssets(registry); err != nil {
@@ -48,14 +50,15 @@ func TestExtractionPromptComposition(t *testing.T) {
promptVersion string promptVersion string
inputs map[string]promptkit.ArtifactRef inputs map[string]promptkit.ArtifactRef
suffixGroups [][]string suffixGroups [][]string
middleSentinel string
}{ }{
{name: "npcs", promptID: npcextract.PromptID, promptVersion: npcextract.SchemaVersion, inputs: commonInputs, suffixGroups: [][]string{{evidenceSentinel}}}, {name: "npcs", promptID: npcextract.PromptID, promptVersion: npcextract.SchemaVersion, inputs: commonInputs, suffixGroups: [][]string{{evidenceSentinel, endpointOrderSentinel}}},
{name: "locations", promptID: locationextract.PromptID, promptVersion: locationextract.SchemaVersion, inputs: commonInputs, suffixGroups: [][]string{{evidenceSentinel}}}, {name: "locations", promptID: locationextract.PromptID, promptVersion: locationextract.SchemaVersion, inputs: commonInputs, suffixGroups: [][]string{{evidenceSentinel, endpointOrderSentinel}}},
{name: "item occurrences", promptID: itemoccurrenceextract.PromptID, promptVersion: itemoccurrenceextract.SchemaVersion, inputs: withPromptInputs(commonInputs, map[string]promptkit.ArtifactRef{ {name: "item occurrences", promptID: itemoccurrenceextract.PromptID, promptVersion: itemoccurrenceextract.SchemaVersion, inputs: withPromptInputs(commonInputs, map[string]promptkit.ArtifactRef{
"item_registry": promptkit.Inline(`{"items":[{"id":"item-registry-sentinel","name":"Torch"}]}`), "item_registry": promptkit.Inline(`{"items":[{"id":"item-registry-sentinel","name":"Torch"}]}`),
}), suffixGroups: [][]string{{evidenceSentinel}, {"item-registry-sentinel"}}}, }), suffixGroups: [][]string{{evidenceSentinel, endpointOrderSentinel}, {"item-registry-sentinel"}}},
{name: "item registry", promptID: itemregistryextract.PromptID, promptVersion: itemregistryextract.SchemaVersion, inputs: commonInputs, suffixGroups: [][]string{{evidenceSentinel}}}, {name: "item registry", promptID: itemregistryextract.PromptID, promptVersion: itemregistryextract.SchemaVersion, inputs: commonInputs, suffixGroups: [][]string{{evidenceSentinel, endpointOrderSentinel}}},
{name: "scene descriptions", promptID: scenedescriptionextract.PromptID, promptVersion: scenedescriptionextract.SchemaVersion, inputs: commonInputs}, {name: "scene descriptions", promptID: scenedescriptionextract.PromptID, promptVersion: scenedescriptionextract.SchemaVersion, inputs: commonInputs, middleSentinel: combatPolicySentinel},
{ {
name: "combat turns", name: "combat turns",
promptID: combatextract.PromptID, promptID: combatextract.PromptID,
@@ -63,7 +66,7 @@ func TestExtractionPromptComposition(t *testing.T) {
inputs: withPromptInputs(commonInputs, map[string]promptkit.ArtifactRef{ inputs: withPromptInputs(commonInputs, map[string]promptkit.ArtifactRef{
"npc_registry": promptkit.Inline(`{"sentinel":"` + npcSentinel + `"}`), "npc_registry": promptkit.Inline(`{"sentinel":"` + npcSentinel + `"}`),
}), }),
suffixGroups: [][]string{{evidenceSentinel}, {npcSentinel}}, suffixGroups: [][]string{{evidenceSentinel, endpointOrderSentinel}, {npcSentinel}},
}, },
{ {
name: "enemy events", name: "enemy events",
@@ -75,7 +78,7 @@ func TestExtractionPromptComposition(t *testing.T) {
"npc_occurrences": promptkit.Inline(`{"sentinel":"npc-occurrences-sentinel"}`), "npc_occurrences": promptkit.Inline(`{"sentinel":"npc-occurrences-sentinel"}`),
}), }),
suffixGroups: [][]string{ suffixGroups: [][]string{
{evidenceSentinel}, {evidenceSentinel, endpointOrderSentinel},
{npcSentinel}, {npcSentinel},
{"combat-turns-sentinel", "npc-occurrences-sentinel"}, {"combat-turns-sentinel", "npc-occurrences-sentinel"},
}, },
@@ -87,7 +90,7 @@ func TestExtractionPromptComposition(t *testing.T) {
inputs: withPromptInputs(commonInputs, map[string]promptkit.ArtifactRef{ inputs: withPromptInputs(commonInputs, map[string]promptkit.ArtifactRef{
"npc_registry": promptkit.Inline(`{"sentinel":"` + npcSentinel + `"}`), "npc_registry": promptkit.Inline(`{"sentinel":"` + npcSentinel + `"}`),
}), }),
suffixGroups: [][]string{{evidenceSentinel}, {npcSentinel}}, suffixGroups: [][]string{{evidenceSentinel, endpointOrderSentinel}, {npcSentinel}},
}, },
{ {
name: "location occurrences", name: "location occurrences",
@@ -96,7 +99,7 @@ func TestExtractionPromptComposition(t *testing.T) {
inputs: withPromptInputs(commonInputs, map[string]promptkit.ArtifactRef{ inputs: withPromptInputs(commonInputs, map[string]promptkit.ArtifactRef{
"location_registry": promptkit.Inline(`{"sentinel":"location-registry-sentinel"}`), "location_registry": promptkit.Inline(`{"sentinel":"location-registry-sentinel"}`),
}), }),
suffixGroups: [][]string{{evidenceSentinel}, {"location-registry-sentinel"}}, suffixGroups: [][]string{{evidenceSentinel, endpointOrderSentinel}, {"location-registry-sentinel"}},
}, },
{ {
name: "spells", name: "spells",
@@ -106,7 +109,7 @@ func TestExtractionPromptComposition(t *testing.T) {
"npc_registry": promptkit.Inline(`{"sentinel":"` + npcSentinel + `"}`), "npc_registry": promptkit.Inline(`{"sentinel":"` + npcSentinel + `"}`),
"spell_catalog": promptkit.Inline(`{"sentinel":"` + catalogSentinel + `"}`), "spell_catalog": promptkit.Inline(`{"sentinel":"` + catalogSentinel + `"}`),
}), }),
suffixGroups: [][]string{{evidenceSentinel}, {npcSentinel}, {catalogSentinel}}, suffixGroups: [][]string{{evidenceSentinel, endpointOrderSentinel}, {npcSentinel}, {catalogSentinel}},
}, },
} }
@@ -155,6 +158,12 @@ func TestExtractionPromptComposition(t *testing.T) {
previousIndex = groupIndex previousIndex = groupIndex
} }
instructionIndex := len(prepared.Messages) - 1 instructionIndex := len(prepared.Messages) - 1
if testCase.middleSentinel != "" {
policyIndex := renderedInputMessageIndex(t, prepared.Messages, testCase.middleSentinel)
if policyIndex <= transcriptIndex || policyIndex >= instructionIndex {
t.Fatalf("combat policy rendered at message %d, want after transcript message %d and before instructions message %d", policyIndex, transcriptIndex, instructionIndex)
}
}
if instructionIndex <= previousIndex { if instructionIndex <= previousIndex {
t.Fatalf("instructions rendered at message %d, want after lane input message %d", instructionIndex, previousIndex) t.Fatalf("instructions rendered at message %d, want after lane input message %d", instructionIndex, previousIndex)
} }

Some files were not shown because too many files have changed in this diff Show More