Compare commits
110 Commits
516af12916
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
| e92bcfa74c | |||
| e6b2ae88d2 | |||
| b2e83bd6e7 | |||
| 7f01c3e79e | |||
| 7b5f4ebd42 | |||
| 4bb4582695 | |||
| 7d3434e5f6 | |||
| 29872b2e28 | |||
| b487d93186 | |||
| 8dd7a4324d | |||
| 04ba87e174 | |||
| a26d6ed042 | |||
| 759d32403f | |||
| 1a7b20c766 | |||
| 85a5b52be7 | |||
| 9d0faabf61 | |||
| 1c3da3e869 | |||
| 9abd93502f | |||
| c8a29a5fa2 | |||
| 916d9210fd | |||
| 3d3f16db4a | |||
| 63c397d86a | |||
| 9a92212632 | |||
| 0ef8931697 | |||
| e00cc45c6b | |||
| ab9b743df6 | |||
| 3bd3c7ebf7 | |||
| 75a3f51cee | |||
| 2be999ebd3 | |||
| 32fe7c5b98 | |||
| 55247c47ab | |||
| 5e5c69bf9d | |||
| b4a81f8b09 | |||
| adfed22e1a | |||
| 75b1e2f68b | |||
| 4473363d9f | |||
| 6eb45e0003 | |||
| 0585ad76dc | |||
| 56145c3b7e | |||
| a478fd86c5 | |||
| 916532100d | |||
| bef8ca263b | |||
| f6d037b613 | |||
| 449b506804 | |||
| 071a78ae22 | |||
| ad1cba41c2 | |||
| 67338798aa | |||
| e95e2f2220 | |||
| 628b8d1800 | |||
| d24d4609b6 | |||
| c7f79fb38e | |||
| 8c071800cf | |||
| 569e12c6f4 | |||
| 5b6eb591b2 | |||
| b630384aa0 | |||
| 297d58f090 | |||
| ee71dc4937 | |||
| 65e5d65d14 | |||
| b40b40aaf3 | |||
| f120be1cb4 | |||
| 17673d74ea | |||
| ef19a03cbf | |||
| 0546f6eb4f | |||
| d28d1062e0 | |||
| b3ebfcef37 | |||
| b70d9f77e3 | |||
| 2a75f40871 | |||
| a705ba74a1 | |||
| 8d9c9e7c87 | |||
| 0b5cc4f251 | |||
| 82ffe85f2d | |||
| b3644abc0e | |||
| d653bf1b90 | |||
| 3e66127b94 | |||
| 5d086c13ca | |||
| d36d4e7689 | |||
| 1456aa51cc | |||
| ffc179c822 | |||
| 8e669a1f14 | |||
| 14bfae216d | |||
| 5a58d87995 | |||
| 557809f364 | |||
| ee600975f0 | |||
| 0d8017e23f | |||
| 37b18edf3d | |||
| cda7a61b47 | |||
| 2ad9283148 | |||
| 41a8a80dda | |||
| 90c7fa6381 | |||
| e3839f8620 | |||
| 0fc2f9ee01 | |||
| 5d6305f21a | |||
| 551e4daea2 | |||
| 3589d33468 | |||
| a22c1a7f59 | |||
| ad85d71b0f | |||
| f3506240c2 | |||
| 70c199aa31 | |||
| e2b82746ab | |||
| 4235507f7b | |||
| b346670cc7 | |||
| 7868c26be7 | |||
| 92e89076a2 | |||
| d9b87347b8 | |||
| 20397ef710 | |||
| fc449863f2 | |||
| 51d62de1f3 | |||
| fc76805075 | |||
| 8e680cf96e | |||
| ece1bca460 |
33
.woodpecker/release.yml
Normal file
33
.woodpecker/release.yml
Normal file
@@ -0,0 +1,33 @@
|
||||
when:
|
||||
- event: tag
|
||||
|
||||
steps:
|
||||
- name: validate-release
|
||||
image: golang:1.25.5
|
||||
commands:
|
||||
- |
|
||||
set -eu
|
||||
|
||||
version="$CI_COMMIT_TAG"
|
||||
release_note="docs/releases/$version.md"
|
||||
|
||||
if ! printf '%s\n' "$version" | grep -E -x 'v(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)' >/dev/null; then
|
||||
printf '%s\n' "invalid release tag: $version" >&2
|
||||
exit 1
|
||||
fi
|
||||
if [ ! -s "$release_note" ]; then
|
||||
printf '%s\n' "missing release note: $release_note" >&2
|
||||
exit 1
|
||||
fi
|
||||
if ! grep -F -x "# Notarius $version" "$release_note" >/dev/null; then
|
||||
printf '%s\n' "release note heading does not match $version" >&2
|
||||
exit 1
|
||||
fi
|
||||
for heading in '## Summary' '## Compatibility' '## Upgrade' '## Changes'; do
|
||||
if ! grep -F -x "$heading" "$release_note" >/dev/null; then
|
||||
printf '%s\n' "release note is missing heading: $heading" >&2
|
||||
exit 1
|
||||
fi
|
||||
done
|
||||
|
||||
./scripts/check-release-source.sh "$version"
|
||||
16
README.md
16
README.md
@@ -28,6 +28,20 @@ For the complete ordered D&D workflow, use
|
||||
[its synthetic transcript](examples/dnd-complete-transcript.json). It
|
||||
demonstrates all implemented D&D lanes and the supporting campaign references.
|
||||
|
||||
## Install A Source Release
|
||||
|
||||
Install a pinned source release with Go:
|
||||
|
||||
~~~
|
||||
GOWORK=off go install \
|
||||
gitea.maximumdirect.net/eric/notarius/cmd/notarius@<tag>
|
||||
~~~
|
||||
|
||||
Replace `<tag>` with a stable release tag such as `vMAJOR.MINOR.PATCH`. The
|
||||
installed command's diagnostic version is described in the [CLI
|
||||
reference](docs/cli.md); maintainers preparing a release should follow [Source
|
||||
Releases](docs/release.md).
|
||||
|
||||
## Documentation
|
||||
|
||||
- [CLI reference](docs/cli.md) — commands, flags, output streams, and exits.
|
||||
@@ -39,6 +53,8 @@ demonstrates all implemented D&D lanes and the supporting campaign references.
|
||||
artifact formats.
|
||||
- [Subprocess consumer guide](docs/consumers/subprocess.md) — invoke Notarius
|
||||
from an orchestrator and consume a published result.
|
||||
- [Complete D&D consumer guide](docs/consumers/dnd-pipeline.md) — run the full
|
||||
D&D pipeline as a subprocess and discover its structured artifacts.
|
||||
- [Internal overview](docs/internal/overview.md) — implemented component map
|
||||
for maintainers.
|
||||
- [Developer guide](docs/development.md) — contributor orientation and
|
||||
|
||||
@@ -42,4 +42,4 @@ output:
|
||||
format: json
|
||||
validation_mode: json_schema
|
||||
schema_path: dnd_combat_turns_llm.v1.json
|
||||
repair_attempts: 0
|
||||
repair_attempts: 1
|
||||
|
||||
@@ -50,4 +50,4 @@ output:
|
||||
format: json
|
||||
validation_mode: json_schema
|
||||
schema_path: dnd_enemy_events_llm.v1.json
|
||||
repair_attempts: 0
|
||||
repair_attempts: 1
|
||||
|
||||
@@ -1,24 +0,0 @@
|
||||
{
|
||||
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
||||
"$id": "notarius.dnd.entity_reconcile.llm",
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["duplicate_groups"],
|
||||
"properties": {
|
||||
"duplicate_groups": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["members", "canonical"],
|
||||
"properties": {
|
||||
"members": {
|
||||
"type": "array",
|
||||
"items": {"type": "string"}
|
||||
},
|
||||
"canonical": {"type": "string"}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -3,10 +3,10 @@ in party possession established by the transcript. This is an occurrence history
|
||||
not an inventory or ledger: do not calculate balances, resolve item identity
|
||||
across records, or infer ownership that the transcript does not establish.
|
||||
|
||||
For every occurrence, copy the exact `item_id` and `name` pair from the supplied
|
||||
item registry. Record a stated quantity as an integer and leave it null when the transcript
|
||||
does not state one. Use a concise observed item name and preserve the stated
|
||||
currency denomination.
|
||||
For every occurrence, use the supplied canonical item `name`. Record a stated
|
||||
quantity as an integer and leave it null when the transcript does not state
|
||||
one. Preserve the stated currency denomination through the selected canonical
|
||||
registry name.
|
||||
|
||||
Use `discovered` when the party learns of or encounters an item without
|
||||
establishing possession. Use `acquired` when the party or a party member gains
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
Use the supplied item registry only to ground each occurrence. Every record
|
||||
must copy one registry item's exact `id` and exact `name`; do not invent,
|
||||
rename, merge, or infer registry items. The registry is not transcript
|
||||
evidence: cite only the current transcript chunk in `source_refs`.
|
||||
must use one registry item's canonical `name`; do not invent, rename, merge,
|
||||
or infer registry items. The registry is not transcript evidence: cite only the
|
||||
current transcript chunk in `source_refs`.
|
||||
|
||||
{{ input "item_registry" }}
|
||||
|
||||
@@ -42,4 +42,4 @@ output:
|
||||
format: json
|
||||
validation_mode: json_schema
|
||||
schema_path: dnd_item_occurrences_llm.v1.json
|
||||
repair_attempts: 0
|
||||
repair_attempts: 1
|
||||
|
||||
@@ -10,9 +10,8 @@
|
||||
"items": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["item_id", "name", "kind", "quantity", "from", "to", "source_refs"],
|
||||
"required": ["name", "kind", "quantity", "from", "to", "source_refs"],
|
||||
"properties": {
|
||||
"item_id": {"type": "string"},
|
||||
"name": {"type": "string"},
|
||||
"kind": {"type": "string"},
|
||||
"quantity": {"type": ["integer", "null"]},
|
||||
@@ -23,10 +22,10 @@
|
||||
"items": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["start_segment", "end_segment"],
|
||||
"required": ["start_unit_id", "end_unit_id"],
|
||||
"properties": {
|
||||
"start_segment": {"type": "integer"},
|
||||
"end_segment": {"type": "integer"}
|
||||
"start_unit_id": {"type": "integer"},
|
||||
"end_unit_id": {"type": "integer"}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -37,4 +37,4 @@ output:
|
||||
format: json
|
||||
validation_mode: json_schema
|
||||
schema_path: dnd_item_registry_llm.v1.json
|
||||
repair_attempts: 0
|
||||
repair_attempts: 1
|
||||
|
||||
@@ -1,8 +1,9 @@
|
||||
Use candidate names and cited transcript windows only to determine whether
|
||||
candidates identify the same item type or unique designation. Do not treat
|
||||
nearby evidence, similar objects, or a shared owner as sufficient. Keep
|
||||
currency denominations, materially different item types, and uncertain aliases
|
||||
separate. Do not infer an item property or uniqueness.
|
||||
Determine whether candidates identify the same item type or unique designation
|
||||
using their contextual labels and cited transcript windows. Do not treat nearby
|
||||
evidence, similar objects, or a shared owner as sufficient.
|
||||
|
||||
Keep currency denominations and materially different item types separate. Keep
|
||||
uncertain aliases separate. Do not infer an item property or uniqueness.
|
||||
|
||||
When selecting a canonical display name, choose one supplied candidate name
|
||||
that is the clearest established designation.
|
||||
|
||||
@@ -12,19 +12,19 @@ messages:
|
||||
- role: system
|
||||
content_file: ./sharedassets/common-dnd-system.md
|
||||
- role: user
|
||||
content_file: ./instructions.md
|
||||
content_file: ./sharedassets/protocol.md
|
||||
- role: user
|
||||
content_file: ./sharedassets/common-dnd-entity-reconciliation.md
|
||||
content_file: ./instructions.md
|
||||
cache_control:
|
||||
type: ephemeral
|
||||
- role: user
|
||||
content_file: ./candidates.md
|
||||
content_file: ./sharedassets/candidates.md
|
||||
- role: user
|
||||
content_file: ./sharedassets/common-dnd-transcript-windows.md
|
||||
content_file: ./sharedassets/transcript-windows.md
|
||||
cache_control:
|
||||
type: ephemeral
|
||||
output:
|
||||
format: json
|
||||
validation_mode: json_schema
|
||||
schema_path: dnd_entity_reconcile_llm.v1.json
|
||||
repair_attempts: 0
|
||||
schema_path: semantic_reconciliation_llm.v1.json
|
||||
repair_attempts: 1
|
||||
|
||||
@@ -20,6 +20,11 @@ location only when the chunk's context supports that coreference. It must not
|
||||
create a registry location, and registry content or provenance must never
|
||||
replace current-chunk evidence.
|
||||
|
||||
For every occurrence, return the exact selector from the location registry:
|
||||
the canonical `name`, plus an empty `registry_refs` array for a unique name or
|
||||
the complete ordered `registry_refs` array for a repeated name. Registry ranges
|
||||
and context identify the location only; they are not occurrence evidence.
|
||||
|
||||
For overlapping support, visited outranks planned, recalled, and mentioned;
|
||||
planned outranks recalled and mentioned; recalled outranks mentioned. A passage
|
||||
may produce multiple records when it independently establishes separate facts,
|
||||
|
||||
@@ -1,9 +1,11 @@
|
||||
A normalized location registry is provided below for identity grounding. It may
|
||||
be empty. Each record contains the exact location ID and canonical display name
|
||||
to copy when the transcript establishes an occurrence of that place.
|
||||
A contextual location registry is provided below for identity grounding. It may
|
||||
be empty. Every record supplies a canonical display name. A name that appears
|
||||
once is selected with that name and an empty `registry_refs` array. A repeated
|
||||
name is selected only by copying both its name and its complete, ordered
|
||||
`registry_refs` array exactly as supplied.
|
||||
|
||||
Registry content is context, not occurrence evidence. Do not derive an
|
||||
occurrence or a source range from the registry, and do not infer a location
|
||||
that is absent from it.
|
||||
occurrence or `source_refs` range from the registry. Do not invent a location
|
||||
or selector that is absent from it.
|
||||
|
||||
{{ input "location_registry" }}
|
||||
|
||||
@@ -42,4 +42,4 @@ output:
|
||||
format: json
|
||||
validation_mode: json_schema
|
||||
schema_path: dnd_location_occurrences_llm.v1.json
|
||||
repair_attempts: 0
|
||||
repair_attempts: 1
|
||||
|
||||
@@ -10,10 +10,21 @@
|
||||
"items": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["location_id", "name", "kind", "source_refs"],
|
||||
"required": ["name", "registry_refs", "kind", "source_refs"],
|
||||
"properties": {
|
||||
"location_id": {"type": "string"},
|
||||
"name": {"type": "string"},
|
||||
"registry_refs": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["start_unit_id", "end_unit_id"],
|
||||
"properties": {
|
||||
"start_unit_id": {"type": "integer", "minimum": 1},
|
||||
"end_unit_id": {"type": "integer", "minimum": 1}
|
||||
}
|
||||
}
|
||||
},
|
||||
"kind": {"enum": ["visited", "planned", "recalled", "mentioned"]},
|
||||
"source_refs": {
|
||||
"type": "array",
|
||||
|
||||
@@ -37,4 +37,4 @@ output:
|
||||
format: json
|
||||
validation_mode: json_schema
|
||||
schema_path: dnd_location_registry_llm.v1.json
|
||||
repair_attempts: 0
|
||||
repair_attempts: 1
|
||||
|
||||
@@ -1,2 +0,0 @@
|
||||
Location candidates:
|
||||
{{ input "candidates" }}
|
||||
@@ -1,6 +1,8 @@
|
||||
Use candidate names and their cited transcript windows to determine whether
|
||||
candidates identify the same physical place. Do not treat matching names,
|
||||
nearby evidence, nested places, or generic labels as sufficient. Keep parent
|
||||
and child places, similarly named places, and uncertain aliases separate.
|
||||
Determine whether candidates identify the same physical place using their
|
||||
contextual labels and cited transcript windows. Do not treat matching names,
|
||||
nearby evidence, nested places, or generic labels as sufficient.
|
||||
|
||||
Keep parent and child places separate, as well as similarly named places and
|
||||
uncertain aliases.
|
||||
|
||||
When selecting a canonical display name, prefer the clearest established name.
|
||||
|
||||
@@ -12,19 +12,19 @@ messages:
|
||||
- role: system
|
||||
content_file: ./sharedassets/common-dnd-system.md
|
||||
- role: user
|
||||
content_file: ./instructions.md
|
||||
content_file: ./sharedassets/protocol.md
|
||||
- role: user
|
||||
content_file: ./sharedassets/common-dnd-entity-reconciliation.md
|
||||
content_file: ./instructions.md
|
||||
cache_control:
|
||||
type: ephemeral
|
||||
- role: user
|
||||
content_file: ./candidates.md
|
||||
content_file: ./sharedassets/candidates.md
|
||||
- role: user
|
||||
content_file: ./sharedassets/common-dnd-transcript-windows.md
|
||||
content_file: ./sharedassets/transcript-windows.md
|
||||
cache_control:
|
||||
type: ephemeral
|
||||
output:
|
||||
format: json
|
||||
validation_mode: json_schema
|
||||
schema_path: dnd_entity_reconcile_llm.v1.json
|
||||
repair_attempts: 0
|
||||
schema_path: semantic_reconciliation_llm.v1.json
|
||||
repair_attempts: 1
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
Extract Dungeons & Dragons NPC occurrences from the supplied
|
||||
transcript. Include an occurrence only when the transcript establishes one
|
||||
supplied NPC, one occurrence kind, and a coherent passage supporting both.
|
||||
Use the exact `npc_id` and matching `name` pair from the supplied NPC registry;
|
||||
never invent an ID or substitute a similar name.
|
||||
Use the supplied canonical NPC `name`; never invent or substitute a similar
|
||||
name. Cite current-transcript evidence for every occurrence.
|
||||
|
||||
Do not summarize, infer relationships, sentiment, factions, motives, aliases,
|
||||
or persistent state. Do not identify player characters, anonymous groups, or
|
||||
|
||||
@@ -42,4 +42,4 @@ output:
|
||||
format: json
|
||||
validation_mode: json_schema
|
||||
schema_path: dnd_npc_occurrences_llm.v1.json
|
||||
repair_attempts: 0
|
||||
repair_attempts: 1
|
||||
|
||||
@@ -10,11 +10,8 @@
|
||||
"items": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["npc_id", "name", "kind", "source_refs"],
|
||||
"required": ["name", "kind", "source_refs"],
|
||||
"properties": {
|
||||
"npc_id": {
|
||||
"type": "string"
|
||||
},
|
||||
"name": {
|
||||
"type": "string"
|
||||
},
|
||||
|
||||
@@ -37,4 +37,4 @@ output:
|
||||
format: json
|
||||
validation_mode: json_schema
|
||||
schema_path: dnd_npc_registry_llm.v1.json
|
||||
repair_attempts: 0
|
||||
repair_attempts: 1
|
||||
|
||||
@@ -1,3 +0,0 @@
|
||||
NPC candidates for identity comparison:
|
||||
|
||||
{{ input "candidates" }}
|
||||
@@ -1,6 +1,7 @@
|
||||
Use candidate aliases and their cited transcript windows to determine whether
|
||||
candidates refer to the same individual. Preserve distinct individuals even
|
||||
when their names are similar.
|
||||
Determine whether candidates refer to the same individual using their
|
||||
contextual labels and cited transcript windows. Preserve distinct individuals
|
||||
even when their names are similar or their contextual descriptions are
|
||||
identical.
|
||||
|
||||
When selecting a canonical display name, prefer a complete, stable proper name
|
||||
over an abbreviation. Prefer an unadorned proper name over that name plus a
|
||||
|
||||
@@ -12,19 +12,19 @@ messages:
|
||||
- role: system
|
||||
content_file: ./sharedassets/common-dnd-system.md
|
||||
- role: user
|
||||
content_file: ./instructions.md
|
||||
content_file: ./sharedassets/protocol.md
|
||||
- role: user
|
||||
content_file: ./sharedassets/common-dnd-entity-reconciliation.md
|
||||
content_file: ./instructions.md
|
||||
cache_control:
|
||||
type: ephemeral
|
||||
- role: user
|
||||
content_file: ./candidates.md
|
||||
content_file: ./sharedassets/candidates.md
|
||||
- role: user
|
||||
content_file: ./sharedassets/common-dnd-transcript-windows.md
|
||||
content_file: ./sharedassets/transcript-windows.md
|
||||
cache_control:
|
||||
type: ephemeral
|
||||
output:
|
||||
format: json
|
||||
validation_mode: json_schema
|
||||
schema_path: dnd_entity_reconcile_llm.v1.json
|
||||
repair_attempts: 0
|
||||
schema_path: semantic_reconciliation_llm.v1.json
|
||||
repair_attempts: 1
|
||||
|
||||
@@ -35,4 +35,4 @@ output:
|
||||
format: json
|
||||
validation_mode: json_schema
|
||||
schema_path: dnd_scene_descriptions_llm.v1.json
|
||||
repair_attempts: 0
|
||||
repair_attempts: 1
|
||||
|
||||
@@ -31,4 +31,4 @@ output:
|
||||
format: json
|
||||
validation_mode: json_schema
|
||||
schema_path: dnd_scenes_llm.v1.json
|
||||
repair_attempts: 0
|
||||
repair_attempts: 1
|
||||
|
||||
@@ -1,6 +0,0 @@
|
||||
Identify only well-supported duplicate groups among the supplied candidates.
|
||||
|
||||
Candidate keys are opaque identifiers. Copy each selected key exactly. A group
|
||||
must contain at least two supplied keys, and its `canonical` key must be one of
|
||||
its members. Do not create keys, records, names, source references, evidence,
|
||||
or replacement values. Omit any uncertain or unsafe group.
|
||||
@@ -1,7 +1,6 @@
|
||||
Transcript units are the only evidence for extracted events and factual claims.
|
||||
Every reported factual claim must be supported by cited transcript units. Use
|
||||
integer `start_unit_id` and `end_unit_id` values from the transcript. Omit
|
||||
`source_id`; Notarius assigns the current source identity.
|
||||
integer `start_unit_id` and `end_unit_id` values from the transcript.
|
||||
|
||||
When supporting evidence is non-contiguous, use multiple narrow ranges rather
|
||||
than a broad range that bridges unrelated conversation.
|
||||
|
||||
@@ -1,7 +1,5 @@
|
||||
You process Dungeons & Dragons gameplay transcripts.
|
||||
|
||||
Rely only on the supplied inputs. They may contain transcription errors,
|
||||
repeated lines, incomplete sentences, and misheard proper nouns.
|
||||
As input, you will receive one or more portions of a transcript. The transcript may contain transcription errors, repeated lines, incomplete sentences, and misheard proper nouns.
|
||||
|
||||
Return exactly one JSON object that conforms to the configured response schema,
|
||||
with no explanatory prose.
|
||||
Return exactly one JSON object that conforms to the configured response schema, with no explanatory prose.
|
||||
|
||||
@@ -1,5 +1,3 @@
|
||||
One extraction chunk from a Dungeons & Dragons gameplay transcript is provided
|
||||
below. Report and infer only what is within this chunk. Its unit IDs retain
|
||||
their source-wide meaning.
|
||||
One extraction chunk from a Dungeons & Dragons gameplay transcript is provided below. Report and infer only what is within this chunk. Its unit IDs retain their source-wide meaning.
|
||||
|
||||
{{ input "transcript" }}
|
||||
|
||||
@@ -1,4 +1,3 @@
|
||||
The complete ordered transcript of this Dungeons & Dragons gameplay session is
|
||||
provided below. It may contain multiple scenes.
|
||||
The complete ordered transcript of this Dungeons & Dragons gameplay session is provided below.
|
||||
|
||||
{{ input "transcript" }}
|
||||
|
||||
@@ -1,6 +0,0 @@
|
||||
Selected Dungeons & Dragons gameplay transcript evidence windows are provided
|
||||
below. They may be incomplete, non-contiguous, or overlapping. Use them to
|
||||
evaluate candidate identity, but do not treat absence outside these windows as
|
||||
evidence.
|
||||
|
||||
{{ input "transcript" }}
|
||||
@@ -47,4 +47,4 @@ output:
|
||||
format: json
|
||||
validation_mode: json_schema
|
||||
schema_path: dnd_spells_llm.v1.json
|
||||
repair_attempts: 0
|
||||
repair_attempts: 1
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
The canonical spell-name catalog for this extraction is provided below as JSON.
|
||||
Return spell names using the catalog's canonical spelling exactly. Aliases and
|
||||
other campaign reference material are not part of this catalog input and must
|
||||
not be copied into the output as spell names.
|
||||
The spell catalog for this extraction is provided below as JSON. Each entry
|
||||
lists a `canonical_name` and its recognized `aliases`. If the transcript uses
|
||||
an alias, select that entry's `canonical_name`. Return spell names using the
|
||||
canonical spelling exactly; never return an alias as a spell name.
|
||||
|
||||
{{ input "spell_catalog" }}
|
||||
|
||||
@@ -1,2 +1,3 @@
|
||||
Item candidates:
|
||||
Candidate material:
|
||||
|
||||
{{ input "candidates" }}
|
||||
@@ -0,0 +1,5 @@
|
||||
Identify only high-confidence duplicate entities among the supplied candidates.
|
||||
|
||||
Preserve distinct entities even when their names are similar. Treat contextual descriptions and transcript evidence as supporting material, not as permission to merge ambiguous records.
|
||||
|
||||
When several records are duplicates, choose as canonical the candidate with the clearest stable identity. Prefer a complete proper name over an abbreviation, and prefer an unadorned proper name over one with incidental descriptors unless the evidence establishes those descriptors as part of the name. A longer name is not inherently more canonical.
|
||||
27
assets/generic/normalize/deduplication/prompts/prompt.yaml
Normal file
27
assets/generic/normalize/deduplication/prompts/prompt.yaml
Normal file
@@ -0,0 +1,27 @@
|
||||
id: generic.semantic_reconciliation
|
||||
version: "v1"
|
||||
inputs:
|
||||
- name: candidates
|
||||
required: true
|
||||
content_type: application/json
|
||||
- name: transcript
|
||||
required: true
|
||||
content_type: application/json
|
||||
messages:
|
||||
- role: system
|
||||
content_file: ./system.md
|
||||
- role: user
|
||||
content_file: ./protocol.md
|
||||
- role: user
|
||||
content_file: ./instructions.md
|
||||
cache_control:
|
||||
type: ephemeral
|
||||
- role: user
|
||||
content_file: ./candidates.md
|
||||
- role: user
|
||||
content_file: ./transcript-windows.md
|
||||
output:
|
||||
format: json
|
||||
validation_mode: json_schema
|
||||
schema_path: semantic_reconciliation_llm.v1.json
|
||||
repair_attempts: 1
|
||||
@@ -0,0 +1,7 @@
|
||||
Use only the positive integer `candidate_id` values supplied in the candidate material.
|
||||
|
||||
Return a duplicate group only when the evidence supports that every selected candidate describes the same underlying entity. Each group must contain at least two distinct candidate IDs, and its `canonical_candidate_id` must be one of those IDs. A candidate may appear in at most one group.
|
||||
|
||||
Omit uncertain matches and candidates that should remain distinct. Do not invent candidates or infer an ID from list position. An empty `duplicate_groups` array is valid.
|
||||
|
||||
The response must conform exactly to the selected JSON schema. Return IDs only: do not copy candidate names, evidence, transcript text, source identifiers, or source ranges into the response.
|
||||
2
assets/generic/normalize/deduplication/prompts/system.md
Normal file
2
assets/generic/normalize/deduplication/prompts/system.md
Normal file
@@ -0,0 +1,2 @@
|
||||
You reconcile structured records that may describe the same underlying entity.
|
||||
Follow the supplied protocol and return only the requested structured result.
|
||||
@@ -0,0 +1,3 @@
|
||||
Transcript evidence windows:
|
||||
|
||||
{{ input "transcript" }}
|
||||
@@ -0,0 +1,32 @@
|
||||
{
|
||||
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
||||
"$id": "notarius.generic.semantic_reconciliation.llm",
|
||||
"title": "notarius_semantic_reconciliation_llm_v1",
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["duplicate_groups"],
|
||||
"properties": {
|
||||
"duplicate_groups": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["candidate_ids", "canonical_candidate_id"],
|
||||
"properties": {
|
||||
"candidate_ids": {
|
||||
"type": "array",
|
||||
"minItems": 2,
|
||||
"items": {
|
||||
"type": "integer",
|
||||
"minimum": 1
|
||||
}
|
||||
},
|
||||
"canonical_candidate_id": {
|
||||
"type": "integer",
|
||||
"minimum": 1
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
# ADR-0012: Resolve opaque entity identifiers deterministically
|
||||
|
||||
**Status:** Accepted
|
||||
**Date:** 2026-08-08
|
||||
|
||||
## Context
|
||||
|
||||
Entity IDs in durable Notarius artifacts are application-owned, deterministic
|
||||
identifiers. They are useful to artifact consumers, but their hash-based form
|
||||
does not help a model distinguish entities and would make the model reproduce
|
||||
an opaque implementation detail. A plain name is likewise insufficient where
|
||||
multiple supplied records share that name.
|
||||
|
||||
The LLM boundary must preserve the typed artifact and durable-schema ownership
|
||||
of [ADR-0003](0003-typed-interfaces-with-two-zone-data-model.md) and the distinction
|
||||
between disambiguating references and source evidence in
|
||||
[ADR-0009](0009-minimal-evidence-grounded-extraction-artifacts.md).
|
||||
|
||||
## Decision
|
||||
|
||||
Callers present a model with semantic selections: a canonical name when it is
|
||||
unique in the request, or a contextual descriptor containing the name and
|
||||
source coordinates when that context is needed to distinguish supplied
|
||||
records. The model returns only those supplied selections. The caller resolves
|
||||
each accepted selection against the request-local supplied records and attaches
|
||||
the opaque application ID deterministically.
|
||||
|
||||
Source coordinates are permitted in a selection solely as identity context.
|
||||
They neither establish an occurrence fact nor replace that occurrence's
|
||||
current-transcript evidence. A selector must resolve exactly; unknown,
|
||||
ambiguous, partial, reordered, or otherwise unsafe selections are not mapped.
|
||||
Where an operation requires a complete grounded artifact, that failure rejects
|
||||
the complete artifact rather than accepting a partially mapped result.
|
||||
|
||||
An explicitly scoped request-local short label is permitted only when a
|
||||
contextual descriptor would be impractical and the caller can deterministically
|
||||
map the label within that one request. Such a label is not a durable ID, must
|
||||
not escape the request boundary, and requires a concrete justification in its
|
||||
own module contract.
|
||||
|
||||
## Alternatives considered
|
||||
|
||||
- Ask the model to return durable IDs. This exposes opaque implementation
|
||||
state, does not improve semantic disambiguation, and makes model output
|
||||
depend on hash formatting.
|
||||
- Select by name alone. This cannot safely distinguish same-name records.
|
||||
- Make request-local labels durable identifiers. This would turn prompt
|
||||
presentation into a public identity contract and create avoidable migration
|
||||
pressure.
|
||||
- Let the model invent identifiers or resolve ambiguity. This makes identity
|
||||
assignment non-deterministic and weakens validation.
|
||||
|
||||
## Consequences
|
||||
|
||||
Durable integration contracts retain their exact ID/name pairs while models
|
||||
operate on readable contextual selections. Calling modules must own selector
|
||||
construction, exact resolution, ambiguity handling, and conversion into their
|
||||
durable artifact type; PromptKit and its adapter remain transport-only.
|
||||
|
||||
Some ambiguous or invalid proposals are deliberately omitted, retried, or
|
||||
rejected according to the caller's existing failure policy. Internal candidate
|
||||
keys may support deterministic request-local mapping, but they are not
|
||||
model-visible selectors or durable data. This adds local validation work while
|
||||
keeping identity assignment auditable and stable.
|
||||
@@ -0,0 +1,89 @@
|
||||
# ADR-0013: Use request-local candidate handles for semantic reconciliation
|
||||
|
||||
**Status:** Accepted
|
||||
**Date:** 2026-08-09
|
||||
|
||||
## Context
|
||||
|
||||
Several typed normalize stage modules need semantic reconciliation after
|
||||
deterministic preprocessing: a model can judge whether source-backed candidates
|
||||
refer to the same underlying entity, while application code remains responsible
|
||||
for constructing the normalized artifact. Requiring the model to reproduce a
|
||||
candidate's full contextual selector makes the response larger and introduces
|
||||
avoidable formatting, ordering, and transcription failure modes.
|
||||
|
||||
Reconciliation must preserve the exact typed artifact boundary established by
|
||||
[ADR-0003](0003-typed-interfaces-with-two-zone-data-model.md), the domain-neutral
|
||||
framework and concrete-domain dependency direction established by
|
||||
[ADR-0004](0004-package-modules-by-domain.md), and the distinction in
|
||||
[ADR-0009](0009-minimal-evidence-grounded-extraction-artifacts.md) between source
|
||||
evidence and auxiliary identity context. It also needs a concrete, narrowly
|
||||
scoped application of the request-local-label exception allowed by
|
||||
[ADR-0012](0012-resolve-opaque-entity-identifiers-deterministically.md).
|
||||
|
||||
## Decision
|
||||
|
||||
Semantic reconciliation will be a domain-neutral framework mechanism used by
|
||||
typed normalize stage modules. A consuming artifact family will retain
|
||||
ownership of its typed records, identity rules, consolidation policy, durable
|
||||
IDs, and domain warnings; the framework mechanism will not infer those rules
|
||||
from arbitrary data.
|
||||
|
||||
For each reconciliation request, deterministic code will assign every eligible
|
||||
model-visible candidate a contiguous, one-based integer handle. The model may
|
||||
receive the candidate's contextual label, source references, and bounded source
|
||||
context needed to judge identity, but its structured response will identify
|
||||
candidates only by those supplied handles. A handle is local to one request,
|
||||
does not represent entity identity, and must never enter a durable artifact or
|
||||
be used to derive a durable ID.
|
||||
|
||||
The model will propose duplicate groups and select one supplied member of each
|
||||
group as canonical. Deterministic code will resolve the handles through the
|
||||
retained request mapping, validate the complete proposal, discard unsafe
|
||||
groups, and apply only validated groups through typed domain-owned policy. The
|
||||
model will not synthesize replacement records or directly mutate an artifact.
|
||||
|
||||
Every reconciliation prompt will combine a mandatory framework-owned protocol
|
||||
and safety policy with an explicitly selected semantic policy. The semantic
|
||||
policy may be the conservative generic policy or a domain-owned policy, but it
|
||||
cannot replace the shared response protocol or deterministic safety boundary.
|
||||
|
||||
## Alternatives considered
|
||||
|
||||
- Return durable application IDs. Opaque IDs do not help semantic judgment,
|
||||
expose application identity mechanics, and make model output reproduce data
|
||||
that deterministic code already owns.
|
||||
- Return names alone or copied contextual selectors. Names can be ambiguous,
|
||||
while reproducing labels and source ranges adds response complexity and
|
||||
creates mismatches without adding semantic information. Request-local
|
||||
handles preserve exact selection without either failure mode.
|
||||
- Ask the model to return synthesized canonical replacement records. This
|
||||
would transfer typed artifact construction, provenance consolidation, and
|
||||
durable identity policy to a probabilistic boundary.
|
||||
- Reconcile reflection-discovered fields or arbitrary JSON. This would weaken
|
||||
the typed artifact contract and move domain semantics into generic code.
|
||||
- Hide reconciliation inside extraction or another stage. This would obscure
|
||||
stage ownership and create cross-stage behavior outside the fixed pipeline;
|
||||
reconciliation remains explicit normalize-stage behavior.
|
||||
- Let each domain replace the complete prompt protocol. This would duplicate
|
||||
safety mechanics and allow domain policy to bypass the common response and
|
||||
validation contract.
|
||||
|
||||
## Consequences
|
||||
|
||||
Model responses become smaller and easier to validate, while deterministic
|
||||
application code retains authority over identity, provenance, ordering, and
|
||||
typed artifact construction. The framework requires a request-local mapping,
|
||||
bounded context preparation, a private integer response contract, proposal
|
||||
assessment, and shared prompt assets. Each consuming artifact family still
|
||||
requires a typed adapter for its irreducibly domain-specific rules.
|
||||
|
||||
Request-local handles are deliberately unsuitable for persistence, logging as
|
||||
entity identity, checkpoint contracts, or cross-request correlation. Changes
|
||||
to shared protocol and policy assets must participate in the normal prompt,
|
||||
schema, and checkpoint fingerprint mechanisms.
|
||||
|
||||
The shared mechanism and its initial D&D registry consumers are now
|
||||
implemented. Current behavior is documented in
|
||||
[Module Internals](../internal/modules.md#semantic-reconciliation) and
|
||||
[D&D Module Internals](../internal/dnd.md#semantic-registry-reconciliation).
|
||||
59
docs/adr/0014-feedback-aware-validation-retries.md
Normal file
59
docs/adr/0014-feedback-aware-validation-retries.md
Normal file
@@ -0,0 +1,59 @@
|
||||
# ADR-0014: Use feedback-aware validation retries
|
||||
|
||||
**Status:** Accepted
|
||||
**Date:** 2026-08-26
|
||||
|
||||
## Context
|
||||
|
||||
Validation can identify a candidate defect after a producer has returned an
|
||||
otherwise well-formed result. Retrying without the validator's deterministic,
|
||||
bounded feedback wastes the useful diagnosis, while treating validator
|
||||
execution failures as defects would ask a producer to repair conditions it
|
||||
cannot control. The mechanism must preserve typed producer ownership,
|
||||
checkpoint safety, and the repository's sensitive-data boundaries.
|
||||
|
||||
## Decision
|
||||
|
||||
The implementation will keep three independent budgets: the producer binding's
|
||||
outer `retries` budget, PromptKit's structured-output repair budget, and each
|
||||
validator's execution-retry budget. Validators will run sequentially in their
|
||||
configured order and aggregate both rejections and execution failures before a
|
||||
candidate disposition is selected.
|
||||
|
||||
A correction-capable producer will provide the exact single LLM response that
|
||||
controlled its candidate using the `single_response_v1` protocol. A correction
|
||||
attempt will reconstruct the ordinary request and append exactly two fresh
|
||||
messages: that latest response as `assistant`, followed by one deterministic
|
||||
aggregate correction request as `user`. Earlier turns will not accumulate.
|
||||
|
||||
Validator failures will not recurse into correction. Pipeline policy owns
|
||||
terminal disposition, with field-by-field producer overrides over pipeline
|
||||
defaults: structural failure and semantic rejection default to `fail_run`, and
|
||||
validator execution failure defaults to `warn_continue`. Validators can report
|
||||
facts and bounded corrective guidance, but never decide disposition.
|
||||
|
||||
Rejected and structurally invalid candidates will not advance. A candidate
|
||||
allowed through after a validator execution failure will retain explicit
|
||||
incomplete-validation provenance and will not be checkpointed. Exact response
|
||||
and correction text remain attempt-local: they are excluded from ordinary
|
||||
errors, warnings, manifests, receipts, caches, checkpoints, and default debug
|
||||
summaries.
|
||||
|
||||
## Alternatives considered
|
||||
|
||||
- Retry every producer after any validation outcome. This conflates producer
|
||||
defects with validator operational failures and wastes retry budget.
|
||||
- Let validators decide whether to continue. This would distribute pipeline
|
||||
disposition policy across validators and undermine consistent defaults.
|
||||
- Reuse the full prior conversation. Accumulated turns introduce unbounded
|
||||
prompt growth and make correction behavior depend on incidental history.
|
||||
- Persist raw responses to simplify diagnosis. Raw model output and correction
|
||||
guidance may be sensitive and do not belong in durable pipeline records.
|
||||
|
||||
## Consequences
|
||||
|
||||
The framework gains transport-neutral correction and candidate contracts,
|
||||
producer capability checks, policy resolution, aggregated validation outcomes,
|
||||
and conservative checkpoint handling. Prompt construction remains inside the
|
||||
LLM adapter, while modules remain responsible for accurately exposing the
|
||||
single response that directly controlled a candidate.
|
||||
12
docs/cli.md
12
docs/cli.md
@@ -10,6 +10,7 @@ defined in [Operations](operations.md).
|
||||
|
||||
~~~
|
||||
notarius help
|
||||
notarius --version
|
||||
notarius run <pipeline-id> --input path/to/source.json [--json] [flags]
|
||||
notarius config validate [--config path/to/config.yml] [--pipeline pipeline-id] [--only lane-a,lane-b]
|
||||
notarius pipelines list [--config path/to/config.yml] [--json]
|
||||
@@ -18,6 +19,17 @@ notarius pipelines list [--config path/to/config.yml] [--json]
|
||||
Running Notarius without arguments, or with **help**, **--help**, or **-h**,
|
||||
writes the command summary to standard output and exits with status 0.
|
||||
|
||||
`notarius --version` is valid only as the sole root argument. It writes exactly
|
||||
`notarius <version>` followed by a newline to standard output and exits with
|
||||
status 0. A tagged `go install` build can report its main-module stable tag,
|
||||
and controlled builds can inject a stable tag at link time through
|
||||
`gitea.maximumdirect.net/eric/notarius/internal/buildinfo.Override`; an ordinary
|
||||
unversioned checkout reports `development`. Invalid injected version content is
|
||||
a runtime error with exit status 1, while extra `--version` arguments are a
|
||||
syntax error with exit status 2. This diagnostic does not replace the
|
||||
[run-result](integrations/run-result.md) or artifact contracts for downstream
|
||||
compatibility decisions.
|
||||
|
||||
## run
|
||||
|
||||
~~~
|
||||
|
||||
@@ -132,7 +132,11 @@ model: example-model
|
||||
Keep credentials out of the local-backend object. A PromptKit profile may name
|
||||
its credential environment variable through `api_key_env`; set that variable
|
||||
only in the run environment. PromptKit owns the
|
||||
[pinned profile-file format](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0/docs/formats.md).
|
||||
[pinned profile-file format](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.9.0/docs/formats.md),
|
||||
including `base_profile` inheritance. Notarius passes profiles through without
|
||||
merging them. Filesystem profiles cannot express PromptKit's in-memory
|
||||
`APIKeyRequired` setting; an unset `api_key_env` is optional and may reach the
|
||||
provider without authorization.
|
||||
The [PromptKit upstream boundary](integrations/pkg-promptkit.md) identifies the
|
||||
supported package API, and [Operations](operations.md#operational-limits)
|
||||
describes the effective concurrency layers.
|
||||
@@ -218,6 +222,8 @@ pipelines:
|
||||
| Field | Type | Default | Rules |
|
||||
| --- | --- | --- | --- |
|
||||
| **llm_profile** | string | none | Optional non-empty default PromptKit profile ID for selected LLM-backed bindings and validators. An explicitly present blank value is invalid. |
|
||||
| **structured_output_repair_attempts** | integer | prompt-owned (1 in maintained production prompts) | Optional structural-repair limit from 0 through 3 for selected LLM-backed bindings and validators. Omission leaves the prompt's declared policy in control; explicit 0 disables structural repair at that scope. |
|
||||
| **validation_policy** | object | see below | Optional terminal policy defaults for producer validation. Its fields inherit independently into chunk, extract, merge, and normalize bindings. |
|
||||
| **input** | module binding | none | Required. |
|
||||
| **chunk** | module binding | **generic** | Optional. |
|
||||
| **output** | module binding | **json** | Optional. |
|
||||
@@ -237,6 +243,42 @@ run-level **--llm-profile** value first, then the binding's **llm_profile**,
|
||||
then the pipeline's **llm_profile**, and finally the PromptKit default.
|
||||
Deterministic bindings do not receive these defaults or run overrides.
|
||||
|
||||
Structural output repair is resolved after module, validator, and `--only` lane
|
||||
selection. An object's **structured_output_repair_attempts** value takes
|
||||
precedence over the pipeline value; otherwise, an LLM-backed binding or
|
||||
validator inherits the pipeline value. If both are omitted, PromptKit uses the
|
||||
prompt's declared repair policy. The value must be an integer from 0 through 3;
|
||||
explicit `null` and non-integer values are invalid. An explicit value on a
|
||||
deterministic binding or validator is invalid, while a pipeline value simply
|
||||
does not apply to deterministic selections.
|
||||
|
||||
`validation_policy` controls terminal disposition for one complete producer
|
||||
attempt and validator chain. It may appear on a pipeline or a **chunk**,
|
||||
**extract**, **merge**, or **normalize** module binding; input, output, and
|
||||
validator bindings reject it. Every field is optional and resolves in binding,
|
||||
pipeline, then application-default order:
|
||||
|
||||
| Field | Values | Default |
|
||||
| --- | --- | --- |
|
||||
| **producer_structural_failure** | **fail_run**, **reject_output** | **fail_run** |
|
||||
| **semantic_rejection** | **fail_run**, **reject_output** | **fail_run** |
|
||||
| **validator_failure** | **warn_continue**, **fail_run** | **warn_continue** |
|
||||
|
||||
The policy object and its fields must be non-null, and unknown fields are
|
||||
rejected. A deterministic producer may not explicitly set
|
||||
**producer_structural_failure** on its binding, although a pipeline-level
|
||||
default remains valid for pipelines that include LLM-backed producers.
|
||||
|
||||
After the producer binding's retry budget is exhausted, an invalid structured
|
||||
response uses **producer_structural_failure**. One or more semantic validator
|
||||
rejections use **semantic_rejection**; rejection takes precedence over an
|
||||
exhausted validator failure or skip. With no rejection, an exhausted validator
|
||||
failure or skip uses **validator_failure**. `reject_output` records the
|
||||
terminal rejection without advancing that candidate. `warn_continue` is valid
|
||||
only for validator execution failure: it advances a structurally valid,
|
||||
otherwise unrejected result with incomplete-validation provenance and without
|
||||
making it reusable checkpoint state.
|
||||
|
||||
A lane has these fields:
|
||||
|
||||
| Field | Type | Default | Rules |
|
||||
@@ -274,17 +316,32 @@ extract:
|
||||
| --- | --- | --- | --- |
|
||||
| **module** | string | none | Required for an object binding. Must be a registered compatible key. |
|
||||
| **llm_profile** | string | none | Optional non-empty PromptKit profile ID for an LLM-backed binding. It overrides the pipeline default unless the run supplies **--llm-profile**. |
|
||||
| **retries** | integer | 0 | Non-negative additional attempts for chunk, extract, merge, and normalize bindings. |
|
||||
| **structured_output_repair_attempts** | integer | pipeline or prompt-owned (1 in maintained production prompts) | Optional structural-repair limit from 0 through 3 for an LLM-backed binding. It overrides the pipeline value; explicit 0 disables structural repair. |
|
||||
| **validation_policy** | object | pipeline or application defaults | Optional field-by-field terminal-policy override for a chunk, extract, merge, or normalize binding. |
|
||||
| **retries** | integer | 0 | Non-negative additional complete producer attempts for chunk, extract, merge, and normalize bindings. This single budget covers operational errors, invalid structured output, module-requested normalization retry, and semantic correction. |
|
||||
| **options** | object | none | Must satisfy the selected module. |
|
||||
| **references** | map | none | Valid only on chunk, extract, merge, and normalize bindings. |
|
||||
| **validators** | list | production chain | Valid only on chunk, extract, merge, and normalize bindings. |
|
||||
|
||||
Omitting **validators** uses the registered chain. **validators: []** selects
|
||||
an empty chain; a non-empty list replaces the chain in the listed order.
|
||||
Validator bindings accept only **module**, **llm_profile**, and **options**.
|
||||
They reject **references**, **retries**, and nested **validators**. Deterministic
|
||||
validators reject an explicit **llm_profile**. Deterministic module bindings
|
||||
also reject an explicit **llm_profile**.
|
||||
Validator bindings accept only **module**, **llm_profile**,
|
||||
**structured_output_repair_attempts**, **retries**, and **options**. Their
|
||||
**retries** value is a non-negative additional validator-execution budget and
|
||||
is valid only when the selected validator is LLM-backed. A validator retry
|
||||
rechecks the same immutable candidate; it never regenerates the producer.
|
||||
They reject
|
||||
**validation_policy**, **references**, and nested **validators**. Deterministic
|
||||
validators reject explicit **llm_profile** and
|
||||
**structured_output_repair_attempts**.
|
||||
Deterministic module bindings also reject those explicit fields.
|
||||
|
||||
An LLM-backed chunk, extract, merge, or normalize producer with both a
|
||||
non-empty validator chain and positive **retries** must declare the supported
|
||||
single-response correction capability. Preparation rejects a configuration
|
||||
that could require semantic correction from a producer that cannot provide an
|
||||
exact prior response. A deterministic producer, or an LLM attempt that did
|
||||
not make a model call, cannot consume a semantic retry after rejection.
|
||||
|
||||
The **json** output module accepts optional **include_chunk_map** and
|
||||
**evidence_context** settings:
|
||||
@@ -319,7 +376,8 @@ Unknown outer or nested option fields are rejected, as are incompatible YAML
|
||||
types. The allowlist remains valid when a run uses lane filtering: a configured
|
||||
lane that is not active for that invocation simply contributes no evidence.
|
||||
Evidence publication is opt-in because it can persist source text and metadata.
|
||||
Its payload contract is [Published Evidence Context](integrations/evidence-context.md).
|
||||
When enabled, it publishes the selected source-unit excerpt defined by the
|
||||
[Published Evidence Context contract](integrations/evidence-context.md).
|
||||
|
||||
## References And Ordered Handoffs
|
||||
|
||||
|
||||
208
docs/consumers/dnd-pipeline.md
Normal file
208
docs/consumers/dnd-pipeline.md
Normal file
@@ -0,0 +1,208 @@
|
||||
# Consuming The Complete D&D Pipeline
|
||||
|
||||
Use this workflow when an orchestrator runs the maintained complete D&D
|
||||
pipeline and consumes its structured JSON artifacts. The generic
|
||||
[subprocess consumer guide](subprocess.md) owns process-level responsibilities;
|
||||
this guide connects that workflow to the complete D&D configuration, its
|
||||
Seriatim input, and its artifact inventory.
|
||||
|
||||
The [CLI reference](../cli.md), [configuration reference](../config.md),
|
||||
[run-result receipt](../integrations/run-result.md), and
|
||||
[published JSON output contract](../integrations/json-output.md) remain the
|
||||
canonical definitions of those public interfaces.
|
||||
|
||||
## Prepare And Validate The Deployment
|
||||
|
||||
Start from the maintained
|
||||
[complete D&D configuration](../../examples/dnd-complete.config.yml). It uses
|
||||
the `dnd-session` pipeline and demonstrates every implemented D&D lane, ordered
|
||||
artifact handoffs, campaign references, chunk-map publication, and evidence
|
||||
context.
|
||||
|
||||
A deployment must provide its own PromptKit profile and campaign reference
|
||||
files. Use absolute paths for service and subprocess deployments. In
|
||||
particular, observe these different resolution rules:
|
||||
|
||||
- reference paths in YAML are resolved relative to the Notarius configuration
|
||||
file; and
|
||||
- `promptkit.profile_file` is resolved relative to the Notarius process working
|
||||
directory.
|
||||
|
||||
Do not copy the repository example's relative profile path into a deployment
|
||||
without also controlling that working directory. The complete path and profile
|
||||
rules are defined in [Configuration](../config.md).
|
||||
|
||||
Preflight the deployed configuration before processing sessions and whenever
|
||||
it changes:
|
||||
|
||||
```sh
|
||||
notarius config validate \
|
||||
--config /absolute/path/to/notarius.yml \
|
||||
--pipeline dnd-session
|
||||
```
|
||||
|
||||
Provide credentials through the environment or the documented configuration
|
||||
mechanism. Do not put credentials in command arguments, generated
|
||||
configuration, or logs.
|
||||
|
||||
## Supply The Transcript
|
||||
|
||||
The complete pipeline consumes a Seriatim JSON document. The
|
||||
[Seriatim input contract](../integrations/seriatim.md) defines its required
|
||||
metadata, segments, and validation rules. Preserve segment IDs: D&D artifact
|
||||
citations use those segment IDs as source-unit ranges.
|
||||
|
||||
When the caller maintains several transcript tiers, use the final trimmed JSON
|
||||
transcript so extraction operates on the same session content presented to
|
||||
later consumers. For example, Narratio identifies this implemented artifact as
|
||||
`narratio.transcript.final_trimmed` and normally stores it at
|
||||
`transcripts/final.trimmed.json`.
|
||||
|
||||
Notarius generates a stable prompt session from the resolved input module and
|
||||
the exact input bytes. An ordinary orchestrator should not pass `--session-id`.
|
||||
Use that override only when intentionally changing the routing relationship
|
||||
between invocations; it is not a credential or output identity.
|
||||
|
||||
## Run Notarius
|
||||
|
||||
Invoke the pipeline with explicit absolute paths and request its
|
||||
machine-readable receipt:
|
||||
|
||||
```sh
|
||||
notarius run dnd-session \
|
||||
--config /absolute/path/to/notarius.yml \
|
||||
--input /absolute/path/to/transcripts/final.trimmed.json \
|
||||
--output-dir /absolute/path/to/notarius-output \
|
||||
--json
|
||||
```
|
||||
|
||||
The caller should:
|
||||
|
||||
- capture stdout and stderr separately;
|
||||
- propagate cancellation and impose an operator-appropriate timeout;
|
||||
- wait for process completion before interpreting stdout; and
|
||||
- retain stderr for diagnosis without copying secrets or transcript content
|
||||
into other logs.
|
||||
|
||||
Only exit status 0 permits decoding stdout as a receipt. Ignore stdout after a
|
||||
nonzero exit because a failed receipt write can leave partial bytes. The
|
||||
[CLI reference](../cli.md#output-streams-and-exit-statuses) defines the complete
|
||||
stream and exit-status contract.
|
||||
|
||||
## Discover The Published Bundle
|
||||
|
||||
Decode the successful stdout document as a supported run-result schema. For
|
||||
the current contract, `schema_version` is `notarius.run-result.v1`. Tolerate
|
||||
unknown fields allowed by that version, but reject an unsupported schema
|
||||
version.
|
||||
|
||||
Use the receipt's absolute `output_directory` as the exact run-specific bundle
|
||||
root. Do not scan the output root for its newest directory, guess a run ID, or
|
||||
construct a bundle path. Resolve `index_file` beneath `output_directory` and
|
||||
reject an absolute logical path or any result that escapes the bundle root.
|
||||
|
||||
The complete configuration uses the application validation defaults. A caller
|
||||
that requires fully validated D&D artifacts must also require receipt
|
||||
`validation_status: approved`; a successful `incomplete` result reflects the
|
||||
configured validator-failure continuation policy and carries its bounded
|
||||
validator provenance in `validation_summaries`.
|
||||
|
||||
Read `index.json` and locate each requested lane in `output_files` by its exact
|
||||
`lane_id`. Do not guess a lane filename. Before decoding a payload:
|
||||
|
||||
1. resolve its descriptor's relative `file` beneath the bundle root with the
|
||||
same confinement check;
|
||||
2. verify the descriptor's media type and schema identity against the linked
|
||||
artifact contract; and
|
||||
3. decode the payload according to that contract.
|
||||
|
||||
The [published JSON output contract](../integrations/json-output.md) defines
|
||||
the index and bundle layout. Treat all paths obtained from a decoded external
|
||||
document as untrusted until confined to their documented root.
|
||||
|
||||
## Complete Artifact Inventory
|
||||
|
||||
When every configured lane is accepted, the complete example publishes these
|
||||
lane artifacts:
|
||||
|
||||
| Lane ID | Purpose | Canonical contract |
|
||||
| --- | --- | --- |
|
||||
| `item-registry` | Canonical registry of encountered items and currency. | [Item registry](../integrations/dnd-item-registry-artifacts.md) |
|
||||
| `npc-registry` | Canonical registry of named NPCs. | [NPC registry](../integrations/dnd-npc-registry-artifacts.md) |
|
||||
| `location-registry` | Canonical registry of named locations. | [Location registry](../integrations/dnd-location-registry-artifacts.md) |
|
||||
| `scene-descriptions` | Classification, title, and summary for each scene. | [Scene descriptions](../integrations/dnd-scene-description-artifacts.md) |
|
||||
| `item-occurrences` | Source-grounded item discovery, acquisition, use, transfer, and loss events. | [Item occurrences](../integrations/dnd-item-occurrence-artifacts.md) |
|
||||
| `spells` | Source-grounded spell casts and casters. | [Spell casts](../integrations/dnd-spell-artifacts.md) |
|
||||
| `combat-turns` | Source-grounded combat turn participation. | [Combat turns](../integrations/dnd-combat-turn-artifacts.md) |
|
||||
| `npc-occurrences` | Source-grounded NPC interaction occurrences. | [NPC occurrences](../integrations/dnd-npc-occurrence-artifacts.md) |
|
||||
| `location-occurrences` | Source-grounded location occurrences. | [Location occurrences](../integrations/dnd-location-occurrence-artifacts.md) |
|
||||
| `enemy-events` | Source-grounded enemy combat events. | [Enemy events](../integrations/dnd-enemy-event-artifacts.md) |
|
||||
|
||||
The JSON encoder always publishes these bundle-management files:
|
||||
|
||||
| File | Purpose |
|
||||
| --- | --- |
|
||||
| `index.json` | Discovery document for lane and pipeline-wide artifacts. |
|
||||
| `manifest.json` | Run provenance and result summaries. |
|
||||
| `rejected.json` | Rejected pipeline outputs. |
|
||||
| `warnings.json` | Accepted-output and run warnings. |
|
||||
|
||||
The complete configuration also requests two pipeline-wide artifacts:
|
||||
|
||||
- [`chunk-map.json`](../integrations/chunk-map.md), the accepted chunk plan and
|
||||
chunk metadata; and
|
||||
- [`evidence-context.json`](../integrations/evidence-context.md), a reading
|
||||
excerpt containing the union of selected cited source units and the
|
||||
configured surrounding window.
|
||||
|
||||
Discover both from their top-level `index.json` descriptors rather than
|
||||
treating them as lanes. Evidence context is convenient reading material, not
|
||||
authoritative provenance; citations in the normalized lane payloads remain the
|
||||
evidence contract.
|
||||
|
||||
Every optional or lane file is published only when its corresponding artifact
|
||||
is available. A successful process does not guarantee that all configured
|
||||
lanes were accepted.
|
||||
|
||||
## Decide What Counts As Consumer Success
|
||||
|
||||
Exit status 0 means Notarius completed the pipeline and published its result
|
||||
bundle. The receipt or bundle may still report warnings, rejected outputs, or
|
||||
missing lane descriptors. A downstream consumer must define its own required
|
||||
artifact set explicitly.
|
||||
|
||||
A caller that claims to consume the complete D&D workflow should normally
|
||||
require all ten lane IDs in the table and verify each descriptor's expected
|
||||
contract. If any required lane is missing, rejected, or incompatible, fail the
|
||||
caller's extraction step while retaining the Notarius bundle for diagnosis. A
|
||||
consumer that needs only a subset may define and document a narrower policy.
|
||||
|
||||
Keep the successful receipt with the complete published bundle. Retain
|
||||
`manifest.json`, `rejected.json`, `warnings.json`, and captured process logs as
|
||||
required by the caller's provenance, diagnosis, and retention policies. Avoid
|
||||
selectively copying payload files without also preserving enough index and
|
||||
manifest information to identify their originating run and contracts.
|
||||
|
||||
The transcript, lane artifacts, evidence context, manifest, debug data, and
|
||||
logs can all contain private campaign information. Apply the same access,
|
||||
publication, and retention controls used for the source transcript.
|
||||
|
||||
## Consumer Checklist
|
||||
|
||||
- Validate the deployed Notarius configuration and `dnd-session` pipeline.
|
||||
- Pass the final trimmed Seriatim JSON transcript with stable segment IDs.
|
||||
- Use absolute configuration, input, output-root, profile, and reference paths
|
||||
in service deployments.
|
||||
- Capture stdout and stderr separately and enforce cancellation and timeout.
|
||||
- Parse stdout only after exit status 0.
|
||||
- Accept only supported receipt, index, and artifact schema versions while
|
||||
tolerating permitted unknown fields.
|
||||
- Use the receipt's `output_directory`; never guess the run directory.
|
||||
- Confine `index_file` and every descriptor path to the published bundle root.
|
||||
- Discover lanes by `lane_id` and verify descriptor compatibility before
|
||||
decoding payloads.
|
||||
- Enforce an explicit required-lane policy and inspect rejections and warnings.
|
||||
- Preserve the receipt and sufficient bundle provenance for every retained
|
||||
artifact.
|
||||
- Protect all transcript-derived files and diagnostic streams as sensitive
|
||||
campaign data.
|
||||
@@ -6,6 +6,10 @@ statuses, while the [run-result receipt](../integrations/run-result.md) and
|
||||
[Published JSON Output contract](../integrations/json-output.md) own the
|
||||
durable result formats.
|
||||
|
||||
For the maintained complete D&D workflow, including its transcript input,
|
||||
configured lane inventory, and downstream acceptance checklist, see
|
||||
[Consuming The Complete D&D Pipeline](dnd-pipeline.md).
|
||||
|
||||
## Run And Check The Process
|
||||
|
||||
Optionally preflight a selected configuration and pipeline before work starts:
|
||||
@@ -55,16 +59,23 @@ contract. The JSON bundle contract links to the available lane contracts.
|
||||
If `index.json` has an `evidence_context` descriptor, treat it as a
|
||||
pipeline-wide artifact rather than a lane entry. Verify its six descriptor
|
||||
fields before decoding the linked file according to the [Published Evidence
|
||||
Context contract](../integrations/evidence-context.md). Use each
|
||||
`evidence_refs` entry as the citation to source material. Its surrounding
|
||||
context range and included units explain the citation, but do not widen or
|
||||
replace the cited source reference.
|
||||
Context contract](../integrations/evidence-context.md). Decode its top-level
|
||||
source-unit array as a reading excerpt. Obtain authoritative citations and lane
|
||||
provenance from the normalized lane artifacts; the excerpt has neither and its
|
||||
nearby units do not widen a lane artifact's cited source reference.
|
||||
|
||||
A zero exit status may still report rejected outputs, warnings, or absent
|
||||
lanes. The caller decides which lane IDs are required for its own work and
|
||||
which are optional; it should make that decision explicitly rather than infer
|
||||
failure from the receipt counts alone.
|
||||
|
||||
When complete validation is required, also require receipt
|
||||
`validation_status: approved` and inspect `validation_summaries`. A successful
|
||||
run with `validation_status: incomplete` contains a structurally valid result
|
||||
that advanced after validator execution could not complete under the configured
|
||||
`warn_continue` policy. It is not reusable checkpoint state and should not be
|
||||
silently treated as fully reviewed by the caller.
|
||||
|
||||
## Preserve Provenance And Handle Data Carefully
|
||||
|
||||
Keep the receipt with the published `manifest.json`, and retain
|
||||
@@ -73,5 +84,5 @@ them. Treat the input, output bundle, cache, debug bundle, and captured process
|
||||
logs as potentially sensitive data. Apply the caller's access controls and
|
||||
retention policy, and avoid copying secrets into arguments, logs, or
|
||||
provenance records. An evidence-context artifact contains source-unit text and
|
||||
metadata, and selected lanes can cover most of an input; preserve and share it
|
||||
only when that source content is authorized for the recipient.
|
||||
metadata and can cover most of an input; preserve and share it only when that
|
||||
source content is authorized for the recipient.
|
||||
|
||||
@@ -18,13 +18,14 @@ implemented component map.
|
||||
| Any documentation addition or revision | [Documentation Policy](policy/documentation.md) | It defines canonical homes, audiences, current-behavior rules, and maintenance requirements. |
|
||||
| Adding, changing, reviewing, or deleting tests | [Testing Policy](policy/testing.md) | It defines risk-based sufficiency, durable test boundaries, test-double guidance, and criteria for retaining tests. |
|
||||
| CLI composition or command behavior | [CLI Internals](internal/cli.md) and [CLI Reference](cli.md) | The internal guide owns composition and command flow; the reference owns public syntax. |
|
||||
| Building a subprocess caller or changing its result protocol | [Subprocess Consumer Guide](consumers/subprocess.md), [Run Result Receipt](integrations/run-result.md), and [CLI Internals](internal/cli.md) | These separate caller workflow, durable receipt contract, and CLI implementation behavior. |
|
||||
| Building a subprocess caller or changing its result protocol | [Subprocess Consumer Guide](consumers/subprocess.md), [Complete D&D Consumer Guide](consumers/dnd-pipeline.md), [Run Result Receipt](integrations/run-result.md), and [CLI Internals](internal/cli.md) | These separate generic caller workflow, the complete D&D workflow, the durable receipt contract, and CLI implementation behavior. |
|
||||
| Configuration loading, resolution, or user-visible configuration behavior | [Configuration Internals](internal/configuration.md) and [Configuration](config.md) | The internal guide owns loading and resolution mechanics; the reference owns the configuration contract. |
|
||||
| Pipeline resolution or execution | [Pipeline Internals](internal/pipeline.md) | It documents profiles, references, validation, retries, checkpoints, and runner behavior. |
|
||||
| Production modules or validators | [Module Internals](internal/modules.md), [D&D Module Internals](internal/dnd.md), and [D&D integration contracts](integrations/) | The generic guide owns extension mechanics, the D&D guide owns shared family conventions, and the contracts own durable output shapes. |
|
||||
| LLM clients, prompts, schemas, profiles, or scheduling | [LLM Runtime](internal/llm.md) | It documents the transport boundary and PromptKit integration. |
|
||||
| Output, cache, resume, or debug artifacts | [Run State Internals](internal/state.md), [Operations](operations.md), and [Configuration](config.md) | These separate implementation details, operator behavior, and configuration contracts. |
|
||||
| External input formats, artifact schemas, or durable output files | [Integration Contracts](integrations/) | Integration documents define external and durable data contracts. |
|
||||
| Release preparation, tagging, publication, or verification | [Source Releases](release.md) and [Documentation Policy](policy/documentation.md) | The release procedure owns maintainer guards and immutable-tag recovery; the policy assigns release-note ownership. |
|
||||
| Proposed or unimplemented behavior | [Roadmap](roadmap/) | Future work belongs only in roadmap documentation until implemented. |
|
||||
|
||||
For an existing subsystem, also inspect its focused tests and the package-local
|
||||
|
||||
@@ -24,12 +24,13 @@ An incompatible shape change requires a new schema version.
|
||||
|
||||
Both extraction and normalization require an `item_registry` reference bound to
|
||||
an earlier normalized `dnd/item-registry` artifact. The registry is immutable
|
||||
for an operation and contributes only its ordered `{id,name}` projection after
|
||||
the shared evidence message. It is never occurrence evidence.
|
||||
for an operation and contributes names-only grounding after the shared evidence
|
||||
message. Notarius resolves the model's selected name into the unchanged exact
|
||||
durable ID/name pair. It is never occurrence evidence.
|
||||
|
||||
Each occurrence must use one exact registry ID/name pair. An extraction response
|
||||
with an unknown ID or mismatched name is rejected as invalid model output; the
|
||||
configured pipeline may retry it and never accepts a partial artifact.
|
||||
with an unknown or ambiguous selected name is rejected as invalid model output;
|
||||
the configured pipeline may retry it and never accepts a partial artifact.
|
||||
Normalization and validation remain defense in depth for artifacts entering
|
||||
through other boundaries: normalization canonicalizes a recognized name by ID,
|
||||
preserves unknown values for the registry validator, and the registry validator
|
||||
|
||||
@@ -85,9 +85,10 @@ for later artifacts.
|
||||
|
||||
`dnd/item-occurrences` requires one approved item registry through its
|
||||
`item_registry` reference slot for both extraction and normalization. Its
|
||||
consumer receives only an ordered, source-free `{id,name}` projection; the
|
||||
registry’s source references are never occurrence evidence. Unknown IDs and
|
||||
mismatched pairs are rejected by the occurrence contract. See the
|
||||
consumer receives names-only grounding; Notarius resolves the selected name
|
||||
into the unchanged exact durable ID/name pair. The registry’s source references
|
||||
are never occurrence evidence. Unknown or ambiguous selections are rejected by
|
||||
the occurrence contract. See the
|
||||
[item-occurrence artifact](dnd-item-occurrence-artifacts.md) for that strict
|
||||
wire contract, [Configuration](../config.md#d-d-reference-slots) for binding
|
||||
rules and validator selection, and the [JSON output contract](json-output.md)
|
||||
|
||||
@@ -73,10 +73,12 @@ complete canonical evidence sequence.
|
||||
|
||||
Both extraction and normalization require exactly one `location_registry` reference of
|
||||
kind `dnd/location-registry`, media type `application/json`, and at most 1 MiB. The
|
||||
registry provides identity grounding only: unknown IDs and mismatched ID/name
|
||||
pairs are rejected rather than guessed or reassigned. The current transcript is
|
||||
the only evidence source for an occurrence; registry evidence and provenance
|
||||
never become occurrence evidence.
|
||||
registry provides identity grounding only. The model selects a supplied
|
||||
contextual name-and-registry-reference descriptor, and Notarius resolves it
|
||||
into the exact durable ID/name pair. Unknown, partial, or ambiguous selections
|
||||
are rejected rather than guessed or reassigned. The current transcript is the
|
||||
only evidence source for an occurrence; registry evidence and provenance never
|
||||
become occurrence evidence.
|
||||
|
||||
See [Configuration](../config.md#d-d-reference-slots) for the selectable slot
|
||||
and generated-handoff compatibility, [D&D module internals](../internal/dnd.md)
|
||||
|
||||
@@ -83,9 +83,11 @@ not evidence for later artifacts.
|
||||
## Consumers and publication
|
||||
|
||||
`dnd/location-occurrences` requires one approved location registry through its
|
||||
`location_registry` reference slot. Its prompt receives an ordered source-free `{id,
|
||||
name}` projection and must not treat registry references as occurrence
|
||||
evidence. See the [location-occurrence artifact](dnd-location-occurrence-artifacts.md)
|
||||
`location_registry` reference slot. Its prompt receives contextual selectors
|
||||
containing a canonical name and registry references; Notarius resolves a
|
||||
selection into the unchanged exact durable ID/name pair. Registry references
|
||||
must not be treated as occurrence evidence. See the
|
||||
[location-occurrence artifact](dnd-location-occurrence-artifacts.md)
|
||||
for that contract, [Configuration](../config.md#references-and-ordered-handoffs)
|
||||
for binding rules, and the [JSON output contract](json-output.md) for
|
||||
publication.
|
||||
|
||||
@@ -67,10 +67,12 @@ for uncertain classification.
|
||||
|
||||
## Identity, evidence, and order
|
||||
|
||||
The required normalized [NPC registry artifact](dnd-npc-registry-artifacts.md) resolves
|
||||
the exact `{npc_id, name}` pair. Unknown IDs and names that do not match their
|
||||
ID are rejected; normalization does not repair names by similarity. Registry
|
||||
references are provenance only and never replace an occurrence's own evidence.
|
||||
The required normalized [NPC registry artifact](dnd-npc-registry-artifacts.md)
|
||||
supplies names-only contextual grounding to the model. Notarius resolves the
|
||||
selected name and writes the exact `{npc_id, name}` pair. An unknown or
|
||||
ambiguous selection rejects the complete model result; normalization does not
|
||||
repair names by similarity. Registry references are provenance only and never
|
||||
replace an occurrence's own evidence.
|
||||
The registry may include an identity established by a factual third-party
|
||||
mention; that provenance alone does not create a `mentioned` occurrence. Each
|
||||
occurrence remains a separately cited fact in the current transcript.
|
||||
|
||||
@@ -82,9 +82,10 @@ with its own cited evidence and category.
|
||||
This registry can ground actor or caster names in the [spell](dnd-spell-artifacts.md)
|
||||
and [combat-turn](dnd-combat-turn-artifacts.md) artifacts. It is required to
|
||||
resolve the canonical `name` in an [NPC occurrence](dnd-npc-occurrence-artifacts.md).
|
||||
Occurrence consumers receive an ordered source-free `{id,name}` projection;
|
||||
spells, combat turns, and the [enemy-event artifact](dnd-enemy-event-artifacts.md)
|
||||
receive names-only grounding for actor or subject display. None of these
|
||||
Occurrence consumers receive names-only grounding; Notarius resolves the
|
||||
selected canonical name and writes the unchanged exact durable ID/name pair.
|
||||
Spells, combat turns, and the [enemy-event artifact](dnd-enemy-event-artifacts.md)
|
||||
also receive names-only grounding for actor or subject display. None of these
|
||||
projections supply later-artifact evidence. [Configuration](../config.md#d-d-reference-slots)
|
||||
owns the `npc_registry` binding rules.
|
||||
The [JSON output contract](json-output.md) defines publication, and
|
||||
|
||||
@@ -67,6 +67,12 @@ including a collision with the embedded catalog. Matching uses the catalog’s
|
||||
case, whitespace, and apostrophe normalization, so authors should avoid names
|
||||
or aliases that normalize to another spell.
|
||||
|
||||
Spell extraction receives the effective catalog as deterministic canonical-name
|
||||
and alias pairs. An alias in the transcript selects its associated canonical
|
||||
name; the extractor is instructed to return that canonical spelling. The
|
||||
projection contains no catalog source metadata or provenance, and aliases
|
||||
remain recognition context rather than transcript evidence.
|
||||
|
||||
The overlay is a recognition aid only. The durable spell-artifact schema and
|
||||
source-evidence rules are defined by the
|
||||
[D&D spell artifact contract](dnd-spell-artifacts.md).
|
||||
|
||||
@@ -1,9 +1,11 @@
|
||||
# Published Evidence Context
|
||||
|
||||
This contract defines the optional `source/evidence-context` artifact emitted
|
||||
by the production JSON output. Its configuration is owned by
|
||||
[Configuration](../config.md#module-bindings-and-validators); its logical-file
|
||||
discovery is owned by [Published JSON Output](json-output.md).
|
||||
by the production JSON output. It is a selected source-unit excerpt for
|
||||
convenient reading alongside normalized lane artifacts; it is not a second
|
||||
citation or provenance model. Its configuration is owned by
|
||||
[Configuration](../config.md#module-bindings-and-validators), and its
|
||||
logical-file discovery is owned by [Published JSON Output](json-output.md).
|
||||
|
||||
## Identity And Discovery
|
||||
|
||||
@@ -26,91 +28,80 @@ its absence means evidence publication was not enabled for that bundle.
|
||||
|
||||
## Payload
|
||||
|
||||
The v1 payload is a JSON object with required `source_id`, `source_digest`,
|
||||
`window_units`, `selected_lanes`, and `contexts` fields. `selected_lanes` and
|
||||
`contexts` are always arrays; an enabled configuration with no accepted direct
|
||||
evidence publishes `contexts: []`.
|
||||
The v1 payload is a top-level JSON array of generic source units. There is no
|
||||
wrapper, source-level metadata, context grouping, lane identifier, or evidence
|
||||
reference in the payload. An enabled configuration with no contributing
|
||||
accepted evidence publishes `[]`.
|
||||
|
||||
```json
|
||||
{
|
||||
"source_id": "session-alpha",
|
||||
"source_digest": "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef",
|
||||
"window_units": 1,
|
||||
"selected_lanes": ["npc_registry", "spells"],
|
||||
"contexts": [
|
||||
{
|
||||
"context_ref": {
|
||||
"source_id": "session-alpha",
|
||||
"start_unit_id": 10,
|
||||
"end_unit_id": 20
|
||||
},
|
||||
"evidence_refs": [
|
||||
{
|
||||
"lane_id": "spells",
|
||||
"source_ref": {
|
||||
"source_id": "session-alpha",
|
||||
"start_unit_id": 10,
|
||||
"end_unit_id": 10
|
||||
}
|
||||
}
|
||||
],
|
||||
"units": [
|
||||
{
|
||||
"id": 10,
|
||||
"kind": "transcript_segment",
|
||||
"text": "Aria casts Cure Wounds.",
|
||||
"ref": {
|
||||
"source_id": "session-alpha",
|
||||
"start_unit_id": 10,
|
||||
"end_unit_id": 10
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 20,
|
||||
"kind": "transcript_segment",
|
||||
"text": "The party regroups.",
|
||||
"ref": {
|
||||
"source_id": "session-alpha",
|
||||
"start_unit_id": 20,
|
||||
"end_unit_id": 20
|
||||
}
|
||||
}
|
||||
]
|
||||
[
|
||||
{
|
||||
"id": 10,
|
||||
"kind": "transcript_segment",
|
||||
"text": "Aria casts Cure Wounds.",
|
||||
"ref": {
|
||||
"source_id": "session-alpha",
|
||||
"start_unit_id": 10,
|
||||
"end_unit_id": 10
|
||||
}
|
||||
]
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 20,
|
||||
"kind": "transcript_segment",
|
||||
"text": "The party regroups.",
|
||||
"ref": {
|
||||
"source_id": "session-alpha",
|
||||
"start_unit_id": 20,
|
||||
"end_unit_id": 20
|
||||
}
|
||||
}
|
||||
]
|
||||
```
|
||||
|
||||
Each context requires a `context_ref` object and `evidence_refs` and `units`
|
||||
arrays. `context_ref` identifies the first and last included unit. Each
|
||||
evidence entry contains a selected `lane_id` and an original `source_ref`. A
|
||||
unit uses the existing source-unit shape: required `id`, `kind`, `text`, and
|
||||
self `ref`, plus optional JSON-object `metadata`. Fixed payload objects reject
|
||||
unknown fields; unit metadata may contain application-defined JSON values.
|
||||
Each source unit has required `id`, `kind`, `text`, and self `ref` fields.
|
||||
`ref` contains `source_id`, `start_unit_id`, and `end_unit_id`, and both unit
|
||||
endpoints identify that unit's `id`. A unit may also contain source-owned
|
||||
`metadata`, an open-ended JSON object. Fixed unit and reference fields are
|
||||
strict: consumers must reject unknown fixed fields, malformed units, invalid
|
||||
self-references, units whose `source_id` differs from other units in the same
|
||||
excerpt, and a payload that is not the array described here.
|
||||
|
||||
## Citations And Context
|
||||
The excerpt preserves each selected unit exactly as represented by the
|
||||
validated generic source document. It does not add evidence-context-specific
|
||||
annotations or reshape source-owned metadata.
|
||||
|
||||
`evidence_refs` are the authoritative citations. They identify the direct
|
||||
references emitted by accepted normalized artifacts. `context_ref` and the
|
||||
units collection include those cited units plus nearby source units selected by
|
||||
the configured window. They are explanatory context, not widened citations.
|
||||
## Selection And Citations
|
||||
|
||||
Only accepted outputs from the configured lane allowlist contribute. Rejected,
|
||||
failed, absent, and lane-filtered outputs do not contribute. The artifact never
|
||||
contains raw input bytes, prompts, model responses, auxiliary reference
|
||||
content, credentials, or filesystem paths.
|
||||
The framework obtains direct source references only through typed evidence
|
||||
projections of accepted normalized artifacts in the configured lane allowlist.
|
||||
It validates each reference against the current source document, expands its
|
||||
range by `window_units` source-unit positions on each side, clamps at document
|
||||
boundaries, and takes the union of all expanded ranges. The output contains
|
||||
each selected source unit once in source-document position order, regardless
|
||||
of numeric unit IDs. Repeated references, overlapping windows, and citations
|
||||
from multiple lanes do not duplicate a unit. Rejected, failed, absent,
|
||||
inactive, and unselected lanes contribute nothing.
|
||||
|
||||
## Ordering And Compatibility
|
||||
Normalized lane artifacts remain authoritative for citations and for which lane
|
||||
cited a range. The excerpt has no lane attribution and must not be used to
|
||||
reconstruct it. Its included nearby units provide reading context only; they
|
||||
do not widen any citation in a lane artifact.
|
||||
|
||||
The selected lane allowlist is lexical. Contexts and units are in source
|
||||
document position order, not numeric unit-ID order. Direct evidence entries
|
||||
are deterministically ordered by lane and source reference. Overlapping or
|
||||
contiguous windows merge, and each source unit appears at most once in the
|
||||
resulting contexts.
|
||||
The excerpt contains at most every generic source unit once. It can therefore
|
||||
equal the complete generic source document when coverage is broad or the
|
||||
window is large. No byte-, token-, or compression-size guarantee is made, and
|
||||
the framework does not truncate the excerpt to meet an arbitrary size limit.
|
||||
|
||||
## Consumer Responsibilities And Data Handling
|
||||
|
||||
The artifact is additive to the JSON bundle and is not a lane payload,
|
||||
normalized-output count, checkpoint, or generated reference. Consumers that
|
||||
do not need it must tolerate the absent optional descriptor. Consumers that do
|
||||
use it should preserve the artifact and its schema identity with the run
|
||||
provenance, and should treat its source text and metadata as sensitive durable
|
||||
content.
|
||||
do not need it must tolerate an absent descriptor. Consumers that do use it
|
||||
should validate the descriptor and payload before use, retain the artifact with
|
||||
its schema identity when needed for a run record, and read citations from the
|
||||
corresponding normalized lane artifacts.
|
||||
|
||||
The excerpt contains source-unit text and source-owned metadata and is durable
|
||||
output. Treat it as sensitive source content, apply appropriate access controls
|
||||
and retention, and do not assume its selected form is materially smaller or
|
||||
less sensitive than the original input.
|
||||
|
||||
@@ -24,7 +24,7 @@ root for the logical discovery described here.
|
||||
| `warnings.json` | Accepted-output and run warnings. |
|
||||
| `lanes/<safe-lane-id>.json` | One normalized artifact payload for each lane. |
|
||||
| `chunk-map.json` | Optional accepted chunk map, when its export is enabled and available. |
|
||||
| `evidence-context.json` | Optional source-context artifact, when evidence publication is enabled. |
|
||||
| `evidence-context.json` | Optional selected source-unit excerpt, when evidence publication is enabled. |
|
||||
|
||||
JSON files are pretty-printed with a trailing newline. Lane payloads are
|
||||
accepted only when their media type is `application/json`.
|
||||
@@ -92,7 +92,7 @@ group into the following externally observable summaries:
|
||||
| Run identity and result | `run_id`, `pipeline_id`, `pipeline_digest`, `schema_version`, `validation_status`, `started_at`, `completed_at` |
|
||||
| Resolved components | `input_module`, `chunker`, `extractors`, `merger`, `normalizer`, `output_encoder`, `artifact_lanes`, `validator_chains`, `module_metadata` |
|
||||
| Source and references | `source_digests`, `references` |
|
||||
| Published result summaries | `normalized_outputs`, `rejected_outputs` |
|
||||
| Published result summaries | `normalized_outputs`, `rejected_outputs`, `validation_summaries` |
|
||||
| Execution summaries | `chunk_plan`, `checkpoint_decisions`, `llm_profiles`, `metadata` |
|
||||
|
||||
`references` records provenance such as the target, slot, origin, digest,
|
||||
@@ -102,6 +102,16 @@ summarize results without embedding lane payload bytes. A chunk-plan summary is
|
||||
provenance for the plan used by this run; cache records, debug artifacts, and
|
||||
other operational state are not published as bundle files.
|
||||
|
||||
Each `validation_summaries` entry is a bounded outcome for one producer result.
|
||||
It has required `status`, `producer_attempt_count`, and `terminal_action`;
|
||||
the stage and affected step, lane, module, or chunk identity are present when
|
||||
applicable. `status` is `complete`, `rejected`, or `incomplete`.
|
||||
`rejecting_validators`, `reason_codes`, and `incomplete_validators` preserve
|
||||
configured validator order and omit later duplicates. Entries contain no raw
|
||||
candidate response, correction guidance, validator diagnostic message, or
|
||||
artifact payload. The same shape may appear as `validation` on an affected
|
||||
rejection entry.
|
||||
|
||||
When present, `metadata.session_id` is the effective non-secret routing
|
||||
correlation identifier used for the run. It can be visible to providers and is
|
||||
not a substitute for a cache or checkpoint identity. Its generation and
|
||||
@@ -127,7 +137,10 @@ distinct even when their profile, provider, and model are otherwise equal.
|
||||
`rejected.json` is always an object with a `rejected` array. Each entry has
|
||||
required `stage` and `message`; `step_id`, `lane_id`, `module_key`, `chunk_id`,
|
||||
`chunk_index`, `validator_name`, `reason_code`, `attempt_count`, and
|
||||
`diagnostic_artifact_path` are present only when applicable.
|
||||
`diagnostic_artifact_path` are present only when applicable. An entry may also
|
||||
contain the bounded `validation` summary described above; the existing singular
|
||||
validator and reason fields remain the first configured rejection for
|
||||
compatibility.
|
||||
|
||||
`warnings.json` is always an object with a `warnings` array. Each warning has
|
||||
`reason_code` and `message`; `scope` is optional. Both arrays are empty when
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
# PromptKit Integration
|
||||
|
||||
Notarius pins
|
||||
[`gitea.maximumdirect.net/eric/promptkit` v0.5.0](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0)
|
||||
[`gitea.maximumdirect.net/eric/promptkit` v0.9.0](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.9.0)
|
||||
as its in-process prompt engine. The upstream
|
||||
[Go package consumer guide](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0/docs/consumers/pkg-promptkit.md)
|
||||
[Go package consumer guide](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.9.0/docs/consumers/pkg-promptkit.md)
|
||||
owns the public engine API, and the upstream
|
||||
[format reference](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0/docs/formats.md)
|
||||
[format reference](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.9.0/docs/formats.md)
|
||||
owns prompt, profile, and schema file contracts.
|
||||
|
||||
## Supported Boundary
|
||||
@@ -15,7 +15,8 @@ Notarius relies on the root `promptkit` package to:
|
||||
- construct an `Engine` with filesystem-backed prompt, schema, and optional
|
||||
operator and application-fallback profile sources;
|
||||
- prepare one frozen execution from a `RunRequest` with named inline artifacts,
|
||||
variables, a direct session ID, prompt identity, and profile selection, then
|
||||
variables, a direct session ID, prompt identity, profile selection, and
|
||||
optional appended rendered messages, then
|
||||
record credential-redacted details and run that exact execution;
|
||||
- return rendered debug material, validated structured output, selected
|
||||
profile, backend, effective model metadata, and token usage;
|
||||
@@ -26,7 +27,7 @@ Notarius relies on the root `promptkit` package to:
|
||||
admission exhaustion through `ErrCapacityExceeded`.
|
||||
|
||||
The pinned
|
||||
[`BackendLocal`, `LocalBackend`, and `WithBackend` API](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0/backends.go)
|
||||
[`BackendLocal`, `LocalBackend`, and `WithBackend` API](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.9.0/backends.go)
|
||||
owns the registration and backend-capacity contract.
|
||||
|
||||
For one completion, the adapter calls `PrepareExecution`, takes a
|
||||
@@ -52,7 +53,7 @@ Notarius sends one stable effective session through PromptKit's direct session
|
||||
field, which is authoritative for provider session behavior. It also retains
|
||||
the same value as the `session_id` prompt variable for maintained prompt
|
||||
compatibility. The generated identifier is 76 ASCII characters, within
|
||||
PromptKit v0.5.0's 256-code-point session limit. Session IDs are non-secret
|
||||
PromptKit v0.9.0's 256-code-point session limit. Session IDs are non-secret
|
||||
correlation identifiers and may be exposed to providers and provider
|
||||
observability. The CLI contract owns generation and override behavior.
|
||||
|
||||
@@ -84,7 +85,39 @@ configuration and deployment workflow are defined in
|
||||
[Configuration](../config.md#promptkit-profiles) and
|
||||
[Operations](../operations.md#promptkit-profile-deployment).
|
||||
|
||||
Notarius supports this boundary against PromptKit v0.5.0. Its fallback source,
|
||||
PromptKit owns `base_profile` resolution under its
|
||||
[pinned format rules](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.9.0/docs/formats.md).
|
||||
Notarius records the selected leaf identity and resolved target without parsing
|
||||
or merging inheritance. An unset filesystem `api_key_env` is optional and may
|
||||
reach the provider without authorization, which can result in a 401 or 403.
|
||||
|
||||
PromptKit v0.9.0 accepts only the `developer`, `system`, `user`, and
|
||||
`assistant` text-chat roles after normalizing case and surrounding whitespace.
|
||||
Maintained Notarius prompt definitions use only `system` and `user`.
|
||||
|
||||
For application-owned semantic correction, Notarius uses PromptKit v0.9.0's
|
||||
`RunRequest.AppendedMessages` after the ordinary rendered prompt. It supplies
|
||||
exactly two messages in order: the latest validated producer response with
|
||||
role `assistant`, then deterministic validation guidance with role `user`.
|
||||
It never exposes a general caller-selected role API, accumulates earlier
|
||||
correction turns, or changes the ordinary prompt prefix. Ordinary requests
|
||||
leave appended messages unset.
|
||||
|
||||
PromptKit preserves supplied content but does not own Notarius's correction
|
||||
bounds. Notarius rejects invalid UTF-8, blank, or oversized assistant material
|
||||
(at most 1 MiB), guidance (at most 64 KiB), and combined content (at most
|
||||
1,114,112 bytes) before preparing the request. The transport-neutral
|
||||
application contract owns defensive copying and these limits. Default request
|
||||
and terminal summaries retain only safe counts, digests, identities, and usage;
|
||||
complete appended messages remain limited to the explicitly requested detailed
|
||||
debug trace.
|
||||
|
||||
PromptKit now obtains its maintained OpenRouter and Rakestrawhome backend and
|
||||
profile catalogs from independently versioned transitive modules. Notarius
|
||||
does not import or register either catalog; PromptKit retains catalog source,
|
||||
identity, precedence, credential, and capacity ownership.
|
||||
|
||||
Notarius supports this boundary against PromptKit v0.9.0. Its fallback source,
|
||||
prepared-execution, inspection, and typed capacity APIs are used as public
|
||||
upstream contracts; other PromptKit APIs or file-format behavior are not
|
||||
implicitly supported. A dependency upgrade requires reviewing the adapter,
|
||||
@@ -97,6 +130,10 @@ pinned upstream documentation.
|
||||
module assets, maps its transport-neutral completion contract, prepares and
|
||||
executes requests, validates output, records provenance, captures debug
|
||||
material, redacts errors, and preserves timeout ownership.
|
||||
|
||||
PromptKit provider error details do not cross the ordinary completion boundary.
|
||||
Notarius exposes a provider-neutral generation category and optional status;
|
||||
redacted provider details are retained only in requested debug material.
|
||||
[D&D Module Internals](../internal/dnd.md) owns the embedded
|
||||
`dnd-extraction` fallback profile and the maintained D&D prompt defaults.
|
||||
[Configuration](../config.md#promptkit-profiles) defines how a Notarius
|
||||
@@ -105,4 +142,7 @@ the conventional local backend.
|
||||
|
||||
PromptKit API or format changes outside this boundary are not implicitly
|
||||
supported. Updating the pinned version requires reviewing the adapter and
|
||||
profile/configuration contracts against the upstream documentation.
|
||||
profile/configuration contracts against the upstream documentation. Maintained
|
||||
production prompts use PromptKit's bounded structural-repair contract; their
|
||||
current declaration is one additional repair attempt. Notarius retains the
|
||||
transport-neutral boundary and does not expose PromptKit types to modules.
|
||||
|
||||
@@ -22,11 +22,15 @@ The current schema version is `notarius.run-result.v1`.
|
||||
| `rejected_output_count` | Yes | Number of recorded rejected outputs. |
|
||||
| `warning_count` | Yes | Number of final run warnings. |
|
||||
| `validation_status` | Yes | The final run manifest validation status. |
|
||||
| `validation_summaries` | No | Bounded per-producer validation outcomes; present when producer work ran. |
|
||||
| `debug_directory` | No | Absolute path to the run-specific debug bundle when requested debug capture completed. |
|
||||
|
||||
For the production `json` output module, `index_file` is present only when the
|
||||
completed run returned exactly one logical output file named `index.json`.
|
||||
For another output module, its absence does not indicate a failed run.
|
||||
`validation_status` is `approved`, `rejected`, or `incomplete`; `incomplete`
|
||||
means one or more otherwise accepted results advanced under validator-failure
|
||||
`warn_continue`.
|
||||
|
||||
```json
|
||||
{
|
||||
@@ -38,7 +42,17 @@ For another output module, its absence does not indicate a failed run.
|
||||
"normalized_output_count": 6,
|
||||
"rejected_output_count": 2,
|
||||
"warning_count": 1,
|
||||
"validation_status": "rejected"
|
||||
"validation_status": "incomplete",
|
||||
"validation_summaries": [
|
||||
{
|
||||
"stage": "extract",
|
||||
"lane_id": "spells",
|
||||
"status": "incomplete",
|
||||
"incomplete_validators": ["dnd/spells/source_refs"],
|
||||
"producer_attempt_count": 1,
|
||||
"terminal_action": "warn_continue"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
@@ -48,8 +62,11 @@ For another output module, its absence does not indicate a failed run.
|
||||
paths. They identify the paths used by Notarius and do not resolve symlinks.
|
||||
`output_directory` is the run-specific bundle, not the configured output root.
|
||||
|
||||
The receipt is a summary and discovery document. It does not contain lane
|
||||
descriptors, payloads, manifest data, rejections, warnings, or file contents.
|
||||
The receipt is a summary and discovery document. Its optional validation
|
||||
summaries contain only stable status, identity, validator names, reason codes,
|
||||
attempt counts, and terminal actions. It does not contain lane descriptors,
|
||||
payloads, manifest payloads, rejection messages, warnings, raw model responses,
|
||||
correction guidance, or file contents.
|
||||
For the production JSON output, resolve `index_file` beneath
|
||||
`output_directory`, reject path escapes, and use the
|
||||
[Published JSON Output contract](json-output.md) to discover logical files and
|
||||
|
||||
@@ -24,10 +24,14 @@ preparation, and runner mechanics after their inputs are supplied.
|
||||
|
||||
## Dispatch And Configuration Handoff
|
||||
|
||||
The root dispatcher handles help, configuration validation, pipeline listing,
|
||||
and a pipeline run. It normalizes injectable options before dispatch so that a
|
||||
missing production dependency fails as a command error rather than reaching
|
||||
execution.
|
||||
The root dispatcher handles help, version reporting, configuration validation,
|
||||
pipeline listing, and a pipeline run. Version reporting resolves build
|
||||
information through `internal/buildinfo` before production composition, so the
|
||||
diagnostic remains available without configuration or runtime collaborators.
|
||||
The public syntax, streams, exit classes, and version semantics are defined by
|
||||
the [CLI reference](../cli.md). Other root commands normalize injectable
|
||||
options before dispatch so that a missing production dependency fails as a
|
||||
command error rather than reaching execution.
|
||||
|
||||
Commands that need configuration use one shared loader. The CLI discovers the
|
||||
file, parses it through **internal/core/config**, starts from defaults, applies
|
||||
|
||||
@@ -74,10 +74,19 @@ profile-free, and no second inheritance decision occurs during execution. The
|
||||
public field definitions and precedence are owned by
|
||||
[Configuration](../config.md#pipelines).
|
||||
|
||||
The resolver retains configured `validation_policy` overrides and derives one
|
||||
detached concrete terminal policy for the chunk producer and every lane's
|
||||
extract, merge, and normalize producers. That field-by-field inheritance is
|
||||
complete before preparation, and the effective values contribute to pipeline
|
||||
and checkpoint identity; execution does not interpret configuration defaults.
|
||||
|
||||
The framework resolver supplies defaults, selects lanes, resolves validator
|
||||
chains, checks registered module and artifact compatibility, validates module
|
||||
options, and returns the fixed ordered pipeline shape. The resulting
|
||||
**EffectiveConfig** retains the selected ID, requested selection and reference
|
||||
options, and returns the fixed ordered pipeline shape. Positive validator retry
|
||||
budgets require an LLM-backed selected validator; deterministic validators are
|
||||
rejected during resolution. Eligible LLM-backed producer specifications also
|
||||
contribute their declared correction protocol to the resolved metadata. The
|
||||
resulting **EffectiveConfig** retains the selected ID, requested selection and reference
|
||||
changes, a clone of the input configuration, and the resolved pipeline.
|
||||
Callers may therefore retain or modify their input slices and maps without
|
||||
changing the resolved result, and later consumers cannot mutate the original
|
||||
@@ -94,8 +103,8 @@ runtime error class described in the [CLI reference](../cli.md#output-streams-an
|
||||
|
||||
The framework assigns the resolved pipeline a deterministic SHA-256 digest
|
||||
after defaults, lane selection, module bindings, reference bindings, validator
|
||||
chains, effective LLM profiles, and artifact schema identity have been
|
||||
resolved. The digest excludes
|
||||
chains, selected correction protocols, effective LLM profiles, and artifact
|
||||
schema identity have been resolved. The digest excludes
|
||||
its own stored value. It identifies resolved composition rather than raw YAML
|
||||
bytes, a debug payload, or all runtime state. The CLI records it as invocation
|
||||
provenance before execution; cache and checkpoint identity have additional
|
||||
|
||||
@@ -33,7 +33,9 @@ typed builder. Scene chunking, every extractor, and NPC, location, and item-regi
|
||||
normalization are registered as `llm_backed`; the remaining current D&D mergers
|
||||
and normalizers are `deterministic`. The metadata is available to catalog inspection and
|
||||
resolved-pipeline debug data and determines which selected bindings inherit the
|
||||
pipeline profile. Configuration remains the canonical owner of the exact keys,
|
||||
pipeline profile. The registry normalizers use `single_response_v1`, forwarding
|
||||
corrections to their reconciliation completion and retaining the accepted raw
|
||||
proposal only as an owned model candidate. Configuration remains the canonical owner of the exact keys,
|
||||
profile precedence, and validator order.
|
||||
|
||||
Private structured-LLM response schemas are deliberately minimal. They reject
|
||||
@@ -42,6 +44,13 @@ unknown fields, while preserving semantic candidates for deterministic
|
||||
validation. Do not promote a private response envelope into a durable schema;
|
||||
the contracts above define durable data.
|
||||
|
||||
A D&D producer that declares `single_response_v1` forwards any supplied
|
||||
semantic correction to its structured completion and returns an owned copy of
|
||||
that completion's exact validated raw response as its model candidate. It does
|
||||
not serialize normalized artifacts to create that candidate, so deterministic
|
||||
identity, evidence, warning, and durable-schema behavior remains separate from
|
||||
the model transport material.
|
||||
|
||||
## Prompt Construction
|
||||
|
||||
D&D LLM-facing content lives beneath `assets/dnd/`. Each module contributes a
|
||||
@@ -84,13 +93,15 @@ replace it with a complete profile of the same ID from the configured PromptKit
|
||||
source. Deployment profile selection is documented in
|
||||
[Configuration](../config.md#promptkit-profiles).
|
||||
|
||||
The transcript assets have distinct consumers. Scene chunking consumes the
|
||||
complete-session `common-dnd-transcript-full.md`; extraction prompts consume
|
||||
the current-chunk `common-dnd-transcript-chunk.md`; and NPC, location, and item
|
||||
normalization consume `common-dnd-transcript-windows.md` alongside their
|
||||
candidate collections. Player, party, glossary, and compatible campaign
|
||||
references provide disambiguating context, not evidence. Reference material is
|
||||
canonically ordered before rendering so equivalent inputs remain stable.
|
||||
The D&D transcript assets have distinct consumers. Scene chunking consumes the
|
||||
complete-session `common-dnd-transcript-full.md`, while extraction prompts
|
||||
consume the current-chunk `common-dnd-transcript-chunk.md`. NPC, location, and
|
||||
item normalization instead mount the generic semantic-reconciliation
|
||||
candidate and transcript-window presentation assets. Player, party, glossary,
|
||||
and compatible campaign references provide disambiguating context only when
|
||||
declared by the active prompt; they never establish evidence. Reference
|
||||
material is canonically ordered before rendering so equivalent inputs remain
|
||||
stable.
|
||||
|
||||
Extraction prompts render the common system and identity messages first, then
|
||||
cached campaign references and the cached chunk transcript. Evidence policy and
|
||||
@@ -100,10 +111,11 @@ reusable extraction prefix identical while preserving the lane-specific suffix.
|
||||
|
||||
Scene chunking intentionally uses a different order: system, cached campaign
|
||||
references, uncached module instructions, then the final ephemeral full
|
||||
transcript. Entity normalization also has its own order: system, uncached
|
||||
module instructions, ephemeral reconciliation policy, uncached candidates, and
|
||||
final ephemeral transcript windows. These orders and cache controls are prompt
|
||||
behavior; change them only through the owning manifest and prompt declaration.
|
||||
transcript. Entity normalization also has its own order: D&D system, mandatory
|
||||
generic protocol, ephemeral domain semantic instructions, generic candidate
|
||||
presentation, and final ephemeral generic transcript windows. These orders and
|
||||
cache controls are prompt behavior; change them only through the owning
|
||||
manifest and prompt declaration.
|
||||
|
||||
## Evidence, Candidates, And Normalization
|
||||
|
||||
@@ -116,25 +128,70 @@ result.
|
||||
|
||||
Default chains keep responsibilities separate: structural validators assess the
|
||||
candidate, source-reference validators resolve cited ranges against the current
|
||||
source, durable-schema validation checks an approved representation, and
|
||||
source and require extraction evidence to stay within the current chunk,
|
||||
durable-schema validation checks an approved representation, and
|
||||
relatedness validators report advisory evidence concerns. The configured order
|
||||
is documented in
|
||||
[Configuration](../config.md#production-validator-keys-and-default-chains).
|
||||
|
||||
Every D&D rejection describes the correction in transcript-grounded domain
|
||||
terms, using contextual names, artifact fields, and source segment ranges when
|
||||
useful. The guidance must not ask the model to reproduce durable entity IDs,
|
||||
hashes, validator module keys, or reason codes. Those identifiers remain in
|
||||
ordinary validation provenance; only the actionable semantic guidance is
|
||||
eligible for the correction prompt.
|
||||
|
||||
Enemy-event extraction additionally rejects a second `engaged` observation for
|
||||
the same comparison identity within one scene-scoped result. Normalization may
|
||||
combine results from distinct scenes, so it intentionally does not apply that
|
||||
rule. Configuration owns the exact validator key and chain position.
|
||||
|
||||
Normalizers are deterministic for spells, combat turns, item occurrences, NPC
|
||||
occurrences, scene descriptions, enemy events, and location occurrences. They canonicalize display
|
||||
values and evidence, use source-document order for stable output, and issue
|
||||
bounded warnings for changes or collapsed duplicates. The NPC and location
|
||||
normalizers are intentional exceptions: each first produces a deterministic
|
||||
candidate set, then may use a bounded structured-LLM proposal to reconcile
|
||||
identity groups. Invalid or unusable proposals retain the deterministic result
|
||||
and surface retry or fallback diagnostics; the model does not directly replace
|
||||
durable records.
|
||||
occurrences, scene descriptions, enemy events, and location occurrences. They
|
||||
canonicalize display values and evidence, use source-document order for stable
|
||||
output, and issue bounded warnings for changes or collapsed duplicates. NPC,
|
||||
item, and location registry normalizers are intentional exceptions: each first
|
||||
produces a deterministic candidate set, then may use a bounded structured-LLM
|
||||
proposal to reconcile identity groups.
|
||||
|
||||
## Semantic Registry Reconciliation
|
||||
|
||||
The three registry normalizers instantiate the domain-neutral
|
||||
`internal/framework/semanticreconcile` engine with default bounds. Each
|
||||
eligible candidate receives a contiguous, one-based `candidate_id` for that
|
||||
request. The model sees that handle, the candidate label and source-free
|
||||
evidence ranges, plus bounded transcript windows; it returns only duplicate
|
||||
groups of supplied handles and one supplied canonical handle per group. It
|
||||
never returns names, evidence, durable IDs, or replacement records. Identical
|
||||
labels and evidence remain independently selectable because their handles are
|
||||
distinct.
|
||||
|
||||
The generic core owns the mandatory handle protocol, candidate and transcript
|
||||
presentation, the private response schema, source-reference validation,
|
||||
candidate and combined-material limits, structured completion, proposal
|
||||
assessment, stable group ordering, and typed plan-application mechanics. The
|
||||
D&D prompt contributes its system message and registry-specific semantic
|
||||
instructions. The generic registrar registers the shared prompt and schema;
|
||||
the D&D registrar registers each consuming prompt and the fallback profile.
|
||||
|
||||
Fewer than two eligible candidates skips the LLM without a semantic warning.
|
||||
An exceeded bound also skips the call and preserves the deterministic
|
||||
preprocessed registry, adding the registry's bounded fallback warning. Invalid
|
||||
structured output or discarded proposal groups use the normalizer's existing
|
||||
retry contract; retry exhaustion preserves the safe deterministic or
|
||||
partially applied result and emits its bounded fallback warning. Provider,
|
||||
transport, cancellation, and context-material failures remain execution
|
||||
errors.
|
||||
|
||||
Application remains typed and registry-owned. All three policies select the
|
||||
canonical member's normalized display name, union member evidence in source
|
||||
order, preserve ungrouped records, and derive durable identity only after
|
||||
consolidation. NPC IDs derive from the final name. Item IDs also derive from
|
||||
the final name, and a typed guard prevents currency aliases from crossing
|
||||
denominations or mixing currency with non-currency records. Location IDs
|
||||
derive from the final name and final evidence, preserving same-name,
|
||||
parent/child, and distinct physical-place identities. Registry warning scopes,
|
||||
reason codes, and postconditions remain outside the generic core.
|
||||
|
||||
## Generated References And Grounding
|
||||
|
||||
@@ -144,11 +201,19 @@ producer provenance; consumers resolve the handed-off artifact into an
|
||||
immutable, validated projection for each operation. External files are checked
|
||||
during preparation, while generated artifacts are resolved at the handoff.
|
||||
|
||||
NPC, location, and item registries project ordered, source-free `{id, name}`
|
||||
pairs to their respective occurrence extractors and normalizers. Exact ID/name
|
||||
matching preserves every identity the registry recognizes, including same-name
|
||||
locations with distinct source anchors. The NPC registry additionally supplies
|
||||
names-only actor grounding to spells, combat turns, and enemy events.
|
||||
NPC and item registry consumers receive names-only grounding. Location
|
||||
consumers receive a contextual selector containing the canonical name and the
|
||||
registry references needed to distinguish same-name places. The calling module
|
||||
resolves those supplied selections locally and maps them into the unchanged
|
||||
durable ID/name pair; an unknown or ambiguous selection rejects the complete
|
||||
occurrence result rather than accepting a partial mapping. The NPC registry
|
||||
additionally supplies names-only actor grounding to spells, combat turns, and
|
||||
enemy events.
|
||||
|
||||
Registry references establish a registry identity and may disambiguate a
|
||||
selection, but never become occurrence evidence. Each occurrence keeps its own
|
||||
current-transcript source references, even when it was grounded through the
|
||||
same registry record.
|
||||
Scene descriptions are eligibility-only projections: they retain current-chunk
|
||||
classification data, not scene prose or evidence, and exist to route combat
|
||||
extraction. Enemy-event extraction also projects combat turns to `actor` and
|
||||
@@ -170,7 +235,7 @@ checkpoint fingerprint.
|
||||
| Spells | May use a spell-catalog overlay and optional NPC grounding; the catalog validator supplies domain-specific semantic checks. |
|
||||
| NPC registry | Establishes transcript-grounded NPC identities, including factual third-party mentions, without assigning occurrence categories. It does not consume an NPC registry, and its normalizer is the LLM-assisted reconciliation exception described above. |
|
||||
| Combat turns | Requires a scene-description artifact. It calls the LLM only for an exact `combat` classification; exact non-combat classifications return an accepted empty result, while missing or mismatched classifications return an empty result with a bounded warning. Optional NPC grounding never becomes evidence. |
|
||||
| Item occurrences | Requires the normalized item registry for exact ID/name grounding at extraction and normalization. Campaign context may disambiguate, but the registry never becomes occurrence evidence. |
|
||||
| Item occurrences | Requires the normalized item registry for exact deterministic grounding at extraction and normalization. Campaign context may disambiguate, but the registry never becomes occurrence evidence. |
|
||||
| Item registry | Produces source-grounded item types and unique designations. Its LLM-assisted reconciliation is proposal-only, preserves distinct currency denominations and item types, and does not create per-instance identities. |
|
||||
| NPC occurrences | Requires the normalized NPC registry at extraction and normalization, using it for canonical actor grounding only. It separately emits cited current-transcript occurrence facts, including `mentioned`, rather than deriving them from registry provenance. |
|
||||
| Scene descriptions | Produces the classifications consumed by combat routing; it does not consume an NPC registry or provide evidence for combat artifacts. |
|
||||
|
||||
@@ -24,6 +24,9 @@ adapter does not own source evidence, artifact conversion, normalization, or
|
||||
durable schemas. Those responsibilities remain with the module and its
|
||||
[integration contract](../integrations/).
|
||||
|
||||
The calling module also resolves contextual entity selections and attaches any
|
||||
application identity; PromptKit and this adapter do not own entity identity.
|
||||
|
||||
`PromptKitClient` validates the request target and prompt identity, maps each
|
||||
named material to a PromptKit inline artifact while preserving its origin URI,
|
||||
passes the supplied request session through to PromptKit's direct per-run
|
||||
@@ -39,6 +42,22 @@ observability. The adapter returns PromptKit’s validated raw bytes rather than
|
||||
re-encoding the decoded target. An empty optional material is represented as
|
||||
one space so its named input is retained by PromptKit.
|
||||
|
||||
When a request includes semantic correction material, the adapter validates and
|
||||
defensively copies it before preparation, then appends exactly two messages
|
||||
after the ordinarily rendered prompt: the prior response as an assistant
|
||||
message and the correction guidance as a user message. Requests without a
|
||||
correction do not add messages or introduce caller roles. Ordinary request
|
||||
summaries record correction byte counts and digests only; complete messages are
|
||||
available solely in an explicitly requested debug trace.
|
||||
|
||||
The adapter leaves the ordinary rendered message prefix, named inputs,
|
||||
variables, session, profile, execution overrides, prepared-execution path, and
|
||||
PromptKit repair policy unchanged for a corrected request. It never imports a
|
||||
PromptKit message type into a module or pipeline contract. PromptKit reports
|
||||
actual structural repair count and cumulative token usage per completion; the
|
||||
pipeline's safe terminal debug record projects those values without copying
|
||||
message content.
|
||||
|
||||
Client construction may also receive a run-wide reasoning-effort override from
|
||||
the CLI factory boundary. The adapter copies the caller-owned pointer and
|
||||
creates a fresh PromptKit execution override for each request: a nil pointer
|
||||
@@ -72,7 +91,7 @@ backend membership as runtime without performing generation. Fallback assets
|
||||
are mounted only when at least one source is registered. The production D&D
|
||||
registrar contributes its `dnd-extraction` fallback, and the maintained D&D
|
||||
prompts select that logical ID by default. PromptKit owns source precedence and
|
||||
profile parsing: an operator-provided matching profile takes precedence over a
|
||||
profile parsing and inheritance: an operator-provided matching profile takes precedence over a
|
||||
fallback profile without Notarius merging either document.
|
||||
When the registration is absent, a profile selecting `backend: local` fails
|
||||
inspection instead of falling back to a built-in or endpoint-only target.
|
||||
@@ -99,7 +118,8 @@ because it changes scheduling rather than execution semantics.
|
||||
Production construction creates one PromptKit client and wraps it in one
|
||||
scheduled client. The scheduler has a fixed, positive permit limit, serves
|
||||
queued calls in FIFO order, and removes a queued call when its context is
|
||||
cancelled. A granted permit is released exactly once on every completion path.
|
||||
cancelled. It rechecks the caller context after admission and before dispatch.
|
||||
A granted permit is released exactly once on every completion path.
|
||||
|
||||
The scheduled wrapper surrounds every `CompleteStructured` call, so concurrent
|
||||
lanes, pipeline retries, and LLM-backed validators share the same provider-call
|
||||
@@ -138,12 +158,28 @@ arrangement and its data-only boundary are defined by
|
||||
[ADR-0011](../adr/0011-centralize-llm-assets.md), rather than by this runtime
|
||||
guide.
|
||||
|
||||
The generic registrar is the sole production registration owner for the
|
||||
semantic-reconciliation default prompt and private response schema. The
|
||||
domain-neutral reconciliation package also exposes only its mandatory protocol
|
||||
and candidate/transcript presentation files for domain prompt manifests. D&D
|
||||
registry normalizers mount those files while retaining ownership and hashing
|
||||
of their D&D system message, semantic instructions, and complete prompt
|
||||
declaration. The response schema is therefore registered once even though
|
||||
several typed normalizers select it.
|
||||
|
||||
Mounted prompt assets determine a module's fingerprint. The fingerprint hashes
|
||||
only the module and shared files explicitly selected by its manifest, so an
|
||||
unrelated asset does not invalidate a checkpoint. Schema loaders validate JSON,
|
||||
attach identity and digest metadata, make defensive copies, and expose
|
||||
diagnostics without raw schema bytes.
|
||||
|
||||
Semantic-reconciliation normalizers extend this identity with the shared
|
||||
response-schema digest, framework policy version, and complete limit-policy
|
||||
digest. Their manifest metadata records the same content-free prompt, schema,
|
||||
policy, and limit identities together with domain identity and normalization
|
||||
policies. Request-local handles, source material, proposal content, and raw
|
||||
asset bytes are not checkpoint metadata.
|
||||
|
||||
Private response schemas validate a model transport envelope. They are not the
|
||||
durable artifact schema and should not be documented as an external wire
|
||||
contract. Durable formats and compatibility rules remain in the
|
||||
@@ -176,7 +212,9 @@ structured-output validation. The adapter reports an empty result, validation
|
||||
failure, empty structured body, or decode failure as
|
||||
`ErrInvalidStructuredOutput`, while retaining the returned raw bytes and debug
|
||||
material when they exist. Provider failures remain operational errors rather
|
||||
than output-validation failures.
|
||||
than output-validation failures. Apart from documented context, capacity, and
|
||||
invalid-output categories, provider error values and types do not cross the
|
||||
adapter error chain; callers receive only a credential-redacted diagnostic.
|
||||
|
||||
When PromptKit rejects backend admission before generation, the adapter maps
|
||||
`promptkit.ErrCapacityExceeded` to
|
||||
@@ -188,13 +226,29 @@ caller context takes precedence. The adapter does not retry capacity failures;
|
||||
the pipeline's existing binding attempt policy sees the operational error and
|
||||
decides whether to rerun the complete operation.
|
||||
|
||||
Prompt-declared repair is executed within PromptKit’s structured-output flow.
|
||||
The current production D&D prompt manifests set repair attempts to zero. That
|
||||
setting does not replace pipeline retry behavior: a binding’s configured retry
|
||||
count reruns its stage attempt after an error or rejection, and an exhausted
|
||||
rejection is a recorded output rather than a provider error. The pipeline owns
|
||||
attempt lifecycle, validation chains, and retry diagnostics; see
|
||||
[Pipeline Internals](pipeline.md#validation-retries-and-output) and the
|
||||
PromptKit executes structural repair within its structured-output flow. The
|
||||
maintained production prompt manifests declare one additional repair attempt.
|
||||
When a resolved binding supplies a repair value, the adapter inspects the
|
||||
prompt, copies its complete output contract, changes only the repair limit, and
|
||||
passes that complete replacement contract to PromptKit. This preserves the
|
||||
prompt's output format, validation mode, schema, and provider structured-output
|
||||
settings.
|
||||
|
||||
A successful repair is an ordinary successful completion, not a warning. The
|
||||
adapter reports PromptKit's actual repair count and its cumulative usage
|
||||
directly, without adding the initial and corrective counts again. Debug prompt
|
||||
material records the configured complete contract; debug response material
|
||||
records the repaired response and actual validation result. If the repair
|
||||
budget is exhausted, the adapter retains the final raw bytes and debug material
|
||||
and reports `ErrInvalidStructuredOutput`. Generation failures during an initial
|
||||
or corrective call remain provider-neutral operational errors with the same
|
||||
redaction boundary.
|
||||
|
||||
Structural repair does not replace pipeline retry behavior: a binding's
|
||||
configured retry count reruns its complete stage attempt after an operational
|
||||
or structural error, module-requested retry, or actionable semantic rejection.
|
||||
The pipeline owns attempt lifecycle, validation chains, and retry diagnostics;
|
||||
see [Pipeline Internals](pipeline.md#validation-retries-and-output) and the
|
||||
[binding reference](../config.md#module-bindings-and-validators).
|
||||
|
||||
## Timeout Ownership
|
||||
@@ -225,6 +279,13 @@ surfaced; when the completion already failed, its call error remains the
|
||||
result. Debug-bundle location, retention, and handling are operational concerns
|
||||
documented in [Operations](../operations.md#debug-bundles).
|
||||
|
||||
The attempt-terminal summary is a separate safe trace record: it contains
|
||||
attempt kinds, validator outcome counts and reason codes, effective policy,
|
||||
terminal action, and repair/usage references. It excludes raw assistant
|
||||
responses and correction text. Those values can appear only in the explicitly
|
||||
requested detailed prompt and response artifacts, which require sensitive-data
|
||||
handling.
|
||||
|
||||
Run manifests receive selected profile summaries, including optional effective
|
||||
backend and reasoning provenance, and component identities—not prompt, schema,
|
||||
source, reference, or response content. The published field semantics belong
|
||||
@@ -234,6 +295,9 @@ redacted before it crosses the runtime boundary. Known-secret redaction is
|
||||
available to other runtime collaborators; it does not make prompt or response
|
||||
contents safe for general logging.
|
||||
|
||||
Generation failures expose an application-owned category and optional HTTP
|
||||
status. Provider code, type, and message remain debug-only, after redaction.
|
||||
|
||||
## Failure Boundaries
|
||||
|
||||
- Construction fails for missing asset registries, mutually exclusive profile
|
||||
|
||||
@@ -22,6 +22,15 @@ to bindings whose declared execution class is `llm_backed` and rejects a
|
||||
binding-specific profile on a deterministic module. The user-facing precedence
|
||||
contract belongs in [Configuration](../config.md#pipelines).
|
||||
|
||||
An eligible LLM-backed chunk, extract, merge, or normalize producer may also
|
||||
declare correction protocol `single_response_v1`. That declaration is a
|
||||
promise that the implementation accepts one attempt-local semantic correction
|
||||
and returns an owned copy of the exact one model response that directly
|
||||
controlled the candidate. It must forward correction only to its structured
|
||||
completion request; it must not manufacture prior-response material by
|
||||
serializing a normalized artifact or expose opaque application IDs. Input,
|
||||
output, validator, and deterministic specs cannot declare the protocol.
|
||||
|
||||
Implementations that accept options must provide both an option validator and
|
||||
a builder. The validator is used while resolving configuration; the builder
|
||||
decodes the same options and constructs the implementation from the prepared
|
||||
@@ -36,20 +45,38 @@ they need, register each leaf implementation, and add any family-owned assets
|
||||
or default validator chains. They return contextual errors so production
|
||||
composition fails at startup rather than at the first run.
|
||||
|
||||
A validator that returns a completed rejection must supply two separate
|
||||
bounded values: a stable `ReasonCode` for provenance and actionable
|
||||
`CorrectionGuidance` for the producer. Guidance identifies the semantic defect
|
||||
and the constraints on one complete corrected replacement. It must not contain
|
||||
validator keys, diagnostic paths, opaque application IDs, or other internal
|
||||
identifiers. An operator-facing `Message` may explain the same event, but the
|
||||
framework never copies it into a model request. Missing or invalid guidance is
|
||||
a validator contract failure.
|
||||
|
||||
An artifact family can register an optional typed evidence projector alongside
|
||||
its codec. The projector returns defensive copies of the artifact's direct
|
||||
generic source references and must use the codec's exact Go type. It does not
|
||||
interpret surrounding context or publish files; the pipeline validates the
|
||||
capability during preparation and the output boundary owns publication. See
|
||||
the [Published Evidence Context contract](../integrations/evidence-context.md)
|
||||
for the durable result.
|
||||
for the durable source-unit excerpt. Lane artifacts retain citation and lane
|
||||
provenance; the framework does not add either to that published excerpt.
|
||||
|
||||
An artifact family is broader than a module: it owns the cohesive domain
|
||||
feature across its artifact type, codec, stage modules, validators, prompt
|
||||
policy, schemas, identity helpers, and reference projections. An extractor and
|
||||
normalizer in one artifact family remain independently registered modules in
|
||||
their respective pipeline stages. This ownership vocabulary does not create a
|
||||
new registry or change the fixed pipeline.
|
||||
|
||||
## Production Composition
|
||||
|
||||
Production composition is intentionally split by family:
|
||||
|
||||
- The generic registrar provides the unit chunker, generic JSON validators,
|
||||
and JSON output encoder.
|
||||
JSON output encoder, and shared semantic-reconciliation prompt and response
|
||||
schema assets.
|
||||
- The Seriatim registrar provides the transcript input adapter. Its external
|
||||
input behavior is defined by the [Seriatim contract](../integrations/seriatim.md).
|
||||
- The D&D registrar provides its codecs, extractors, mergers, normalizers,
|
||||
@@ -60,6 +87,42 @@ The CLI owns the composition that invokes these registrars. A module package
|
||||
may register its own family but must not assemble the CLI or make framework
|
||||
packages depend on production extensions.
|
||||
|
||||
## Semantic Reconciliation
|
||||
|
||||
`internal/framework/semanticreconcile` is a domain-neutral strategy used by a
|
||||
typed normalize module; it is not itself a selectable stage module. A
|
||||
source-backed artifact-family normalizer projects its deterministic records
|
||||
into contextual candidates and owned typed record envelopes, supplies its
|
||||
chosen prompt identity and resolved LLM profile, and constructs an engine with
|
||||
explicit limits. The core filters invalid evidence, assigns contiguous
|
||||
request-local integer handles, renders bounded candidate and transcript
|
||||
materials, invokes the structured-completion boundary, and assesses the
|
||||
returned duplicate groups into a stable non-overlapping plan.
|
||||
|
||||
The normalizer then applies that plan through a typed `ApplicationPolicy`. The
|
||||
core preserves ungrouped records, contribution order, and provenance while the
|
||||
artifact family owns group guards, field and evidence consolidation, durable
|
||||
ID derivation, retry and fallback presentation, warnings, and postconditions.
|
||||
Request-local handles do not enter the typed value or durable artifact. Fewer
|
||||
than two eligible candidates skips model invocation; exceeding a candidate or
|
||||
combined-material bound preserves the deterministic result under the family's
|
||||
fallback policy. Provider, transport, cancellation, and context-construction
|
||||
failures remain execution errors.
|
||||
|
||||
When the engine actually makes a proposal call, its typed result carries the
|
||||
owned exact proposal response under the same correction contract as other
|
||||
eligible producers. Deterministic skip, limit, and fallback outcomes carry no
|
||||
model candidate, so a later rejection applies terminal policy without spending
|
||||
an ineffective semantic retry.
|
||||
|
||||
The core supplies a conservative generic prompt and the single private
|
||||
response schema. A domain prompt may substitute its semantic instructions but
|
||||
mounts the core-owned protocol and candidate/transcript presentation assets.
|
||||
Prompt, schema, policy, and limit identities participate in manifest metadata
|
||||
and checkpoint fingerprints. The generic registrar owns production
|
||||
registration of those shared assets; a consuming domain registrar owns only
|
||||
its domain prompt.
|
||||
|
||||
## Adding Or Changing A Module
|
||||
|
||||
1. Choose the pipeline stage and the typed artifact boundary. Put external
|
||||
@@ -72,9 +135,13 @@ packages depend on production extensions.
|
||||
3. Implement strict option decoding, construction, and the typed stage
|
||||
interface. Preserve caller ownership: do not retain mutable request data
|
||||
and return defensive copies where an implementation exposes stored data.
|
||||
If declaring correction capability, forward the request correction and
|
||||
retain only the exact validated response that controlled the result.
|
||||
4. Register the module through its typed registry helper and add it to the
|
||||
owning family registrar. Add a default validator chain only when that
|
||||
family owns the behavior; otherwise require an explicit compatible chain.
|
||||
Every rejection path in a validator must provide actionable correction
|
||||
guidance while retaining its stable internal reason code.
|
||||
5. Update the selectable-key and chain reference in
|
||||
[Configuration](../config.md#production-module-keys), the applicable
|
||||
integration contract, and focused tests. Keep the configuration document
|
||||
|
||||
@@ -25,10 +25,12 @@ physical state roots.
|
||||
| Area | Implemented owners | Responsibility |
|
||||
| --- | --- | --- |
|
||||
| Executable and command boundary | **cmd/notarius**, **internal/cli** | Process entry, command dispatch, configuration discovery, production composition, runtime collaborator setup, durable file placement, and user-facing reporting. |
|
||||
| Build information | **internal/buildinfo** | Resolves a stable linked release tag or build metadata for the diagnostic root version command. |
|
||||
| Configuration | **internal/core/config** | Defaults, strict YAML parsing, environment overrides, structural validation, effective resolution, redaction, and resolved-composition summaries. |
|
||||
| Generic models | **internal/core/source**, **internal/core/artifacts**, **internal/framework/contracts** | Source documents and chunks, manifests and provenance, plus typed artifact, reference, validation, output, and structured-completion contracts. |
|
||||
| Pipeline framework | **internal/framework/pipeline** | Registries, profile and reference resolution, typed preparation, validation, retry coordination, ordered execution, handoff, and result assembly. |
|
||||
| LLM and prompt runtime | **internal/framework/llm**, **internal/framework/promptfs** | Provider-neutral structured completions, scheduling, profile recording, prompt assets, schema registration, and credential-shaped-value redaction. |
|
||||
| Semantic reconciliation | **internal/framework/semanticreconcile** | Bounded source-backed candidate preparation, request-local handle proposals, deterministic assessment, typed plan application, and reconciliation identity metadata; see [Module Internals](modules.md#semantic-reconciliation) and [D&D Module Internals](dnd.md#semantic-registry-reconciliation). |
|
||||
| Embedded LLM content | **assets** | Read-only centralized LLM-facing content, scoped by its consuming package; see [LLM Runtime](llm.md#prompt-and-schema-assets) and [D&D Module Internals](dnd.md#prompt-construction). |
|
||||
| Runtime state | **internal/core/fileio**, **internal/core/debugbundle**, **internal/framework/checkpoint**, **internal/framework/chunkplan**, **internal/framework/chunkmap**, **internal/framework/debug** | Confined atomic files, debug bundles, checkpoint and chunk-plan state, accepted chunk maps, and pipeline-facing debug recording. |
|
||||
| Production extensions | **internal/modules/generic**, **internal/modules/seriatim**, **internal/modules/dnd** | Domain-neutral extensions, Seriatim input support, and D&D extraction families registered into the production catalog. |
|
||||
@@ -49,8 +51,9 @@ the CLI composition boundary.
|
||||
composition, and path safety.
|
||||
- [LLM Runtime](llm.md): structured completion, scheduling, prompt assets,
|
||||
profiles, and secret handling.
|
||||
- [Module Internals](modules.md): generic extension registration, module
|
||||
construction, validation, and reference mechanics.
|
||||
- [Module Internals](modules.md): generic extension registration, artifact
|
||||
families, module construction, semantic reconciliation, validation, and
|
||||
reference mechanics.
|
||||
- [D&D Module Internals](dnd.md): shared D&D extractor conventions, generated
|
||||
reference projections, and lane-specific exceptions. Durable D&D and
|
||||
Seriatim data shapes remain in the [integration contracts](../integrations/).
|
||||
|
||||
@@ -32,13 +32,26 @@ Resolution turns a configured pipeline profile into a **ResolvedPipeline**.
|
||||
It normalizes the pipeline and lane identities, applies stage defaults, selects
|
||||
requested lanes where that is supported, resolves validator chains, checks
|
||||
module capabilities and typed artifact compatibility, validates options, and
|
||||
assigns a deterministic resolved-composition digest. The resolved pipeline
|
||||
contains bindings and declared reference targets, not external reference bytes.
|
||||
assigns a deterministic resolved-composition digest. A correction protocol is
|
||||
selected from each eligible LLM-backed producer specification and becomes part
|
||||
of that resolved identity; only `single_response_v1` is currently supported.
|
||||
Preparation rejects an LLM-backed producer that combines a non-empty validator
|
||||
chain with positive producer retries unless it declares that protocol. Producers
|
||||
without validators or without retries remain valid without correction support.
|
||||
The resolved pipeline contains bindings and declared reference targets, not
|
||||
external reference bytes.
|
||||
After selection, the resolver applies command, binding, and pipeline profile
|
||||
precedence to LLM-backed bindings and validators only; prompt defaults remain
|
||||
an empty resolved binding profile. Deterministic bindings remain profile-free.
|
||||
These effective values are part of the digest, so execution and checkpoint
|
||||
consumers do not repeat profile inheritance.
|
||||
an empty resolved binding profile. It resolves structural output repair
|
||||
separately: a binding's `structured_output_repair_attempts` value wins, then a
|
||||
pipeline value applies to LLM-backed bindings and validators, and omission
|
||||
leaves the prompt-owned policy intact. An explicit repair value on a
|
||||
deterministic binding is rejected. Resolved bindings own copied repair values,
|
||||
and these effective values are part of the digest, so execution and checkpoint
|
||||
consumers do not repeat profile inheritance or configuration resolution.
|
||||
Each LLM request receives its own copy of that resolved value. PromptKit spends
|
||||
it only for structural correction inside one completion; the runner's binding
|
||||
retry policy remains the separate outer budget for complete stage attempts.
|
||||
Configuration resolution supplies the selected profile and catalog; see
|
||||
[Configuration Internals](configuration.md).
|
||||
|
||||
@@ -46,16 +59,20 @@ External reference materialization happens before preparation. The materializer
|
||||
checks that each slot is declared by the selected module, resolves a file path
|
||||
relative to the correct configuration or working-directory origin, reads
|
||||
UTF-8 text, verifies media type and size limits, and retains bounded
|
||||
provenance. A generated-artifact selector remains declared but has no bytes
|
||||
until its producing step completes.
|
||||
provenance. For a positive slot limit, it reads at most the limit plus one byte
|
||||
and rejects overflow before retaining content. A generated-artifact selector
|
||||
remains declared but has no bytes until its producing step completes.
|
||||
|
||||
Preparation is the construction boundary. It validates the resolved shape and
|
||||
registry set, clones the resolved data, then constructs the input adapter,
|
||||
chunker, stage-local validators, every typed lane, and output encoder with
|
||||
cloned options, references, and shared dependencies. It also collects stable
|
||||
checkpoint fingerprints. Missing registrations, incompatible typed entries,
|
||||
nil implementations, and constructor failures are reported before source
|
||||
parsing or any stage operation begins.
|
||||
chunker, stage-local validators, every typed lane, and output encoder. The
|
||||
prepared producer metadata preserves each selected correction protocol, and
|
||||
the resolved digest carrying that metadata participates in checkpoint identity.
|
||||
Each registered builder receives its own cloned build request immediately
|
||||
before its module-owned code runs. Preparation also collects stable checkpoint
|
||||
fingerprints. Missing registrations, incompatible typed entries, nil
|
||||
implementations, and constructor failures are reported before source parsing
|
||||
or any stage operation begins.
|
||||
|
||||
An output encoder can opt into source-evidence publication through its output
|
||||
policy. Preparation keeps the configured lane allowlist and active lanes
|
||||
@@ -115,15 +132,49 @@ for started workers, and prevents output encoding.
|
||||
|
||||
Every chunk, extract, merge, and normalize candidate passes its resolved
|
||||
validator chain. Validators receive immutable canonical input appropriate to
|
||||
their target: chunks, typed values, or serialized codec bytes. They may
|
||||
approve, approve with warnings, reject, or fail. A rejection is an ordinary
|
||||
pipeline result; a validator error is a framework error.
|
||||
their target: chunks, codec-decoded typed candidates, or serialized codec
|
||||
bytes. Each typed validator receives a newly decoded value from the one
|
||||
candidate serialization for that attempt, while serialized validators receive
|
||||
separately owned representation bytes and schema metadata. They may approve,
|
||||
approve with warnings, reject, fail, or be skipped when a runtime prerequisite
|
||||
is unavailable. The shared executor settles every configured validator in
|
||||
order. A failed LLM-backed validator retries only itself against the same
|
||||
immutable candidate; it does not regenerate the producer or alter the
|
||||
validator request. Rejections stop that validator, while other configured
|
||||
validators still run. The executor retains ordered results, bounded
|
||||
deduplicated correction guidance from rejections, and only the final exhausted
|
||||
failure outcome for each validator. The correction builder keeps first
|
||||
occurrence order, omits internal reason codes, validator names, and operator
|
||||
messages, and requests one complete replacement. Missing guidance or an
|
||||
oversized aggregate is a framework contract error; guidance is never inferred
|
||||
or truncated.
|
||||
|
||||
The runner applies the binding's retry policy around a stage operation and its
|
||||
complete validation chain. It preserves warnings only from the final accepted
|
||||
or rejected attempt. Cancellation stops retries. Normalizer-specific retry
|
||||
directives consume this same budget and validate any final safe fallback through
|
||||
the normalizer chain.
|
||||
or rejected attempt, plus one fixed warning per validator whose execution
|
||||
budget was exhausted under `warn_continue`. Cancellation stops retries.
|
||||
Normalizer-specific retry directives consume this same budget and validate any
|
||||
final safe fallback through the normalizer chain.
|
||||
|
||||
The artifact-neutral producer-attempt state machine owns that shared budget,
|
||||
attempt provenance, semantic-correction material, and terminal-policy
|
||||
selection. It accepts producer and complete-validation closures, so artifact
|
||||
materialization, cache handling, checkpoints, and debug output stay at the
|
||||
operation boundary. It distinguishes operational, structural, module-requested,
|
||||
and semantic retries. A semantic retry is available only for a valid latest
|
||||
`single_response_v1` candidate; a deterministic or no-model rejection instead
|
||||
settles the semantic policy immediately. Structural-output errors alone use the
|
||||
structural policy, and validation failure without rejection settles the
|
||||
validator-failure policy without regenerating the producer.
|
||||
|
||||
Chunk planning uses this state machine for generated plans. A rejected or
|
||||
validation-incomplete automatic cache hit is not model material and therefore
|
||||
falls through to a fresh initial generation at producer attempt one; it neither
|
||||
receives a correction, consumes retry budget, promotes cached-candidate
|
||||
warnings, nor overwrites the stored record. An incomplete cache validation
|
||||
under `fail_run` terminates instead. Only a newly generated, completely
|
||||
validated plan is published to the chunk-plan store. Rejected plans never
|
||||
advance, and validation-incomplete plans remain unpublishable.
|
||||
|
||||
After terminal lane work, the runner assembles manifest provenance, normalized
|
||||
artifacts, rejections, warnings, and an optional accepted chunk map. When an
|
||||
@@ -137,6 +188,15 @@ The CLI publishes those files only after the runner returns without a framework
|
||||
error. Logical file names and schemas are defined by the [output integration
|
||||
contracts](../integrations/).
|
||||
|
||||
For every completed producer disposition, the runner projects one bounded
|
||||
validation summary to the manifest, the affected rejection when present, and
|
||||
the CLI result receipt. The summary records status, configured-order rejecting
|
||||
validators and reason codes, incomplete validators, producer-attempt count,
|
||||
and terminal action. It contains no operator message, correction guidance, or
|
||||
model response. `complete`, `rejected`, and `incomplete` describe the final
|
||||
candidate disposition; a run-level `incomplete` status indicates at least one
|
||||
current-run output advanced under `warn_continue`.
|
||||
|
||||
## Checkpoint And Debug Hooks
|
||||
|
||||
The runner receives checkpoint and debug interfaces rather than roots. It
|
||||
@@ -146,6 +206,20 @@ handoff. Generated-reference dependencies participate in checkpoint decisions.
|
||||
Selective recomputation can require a canonical accepted normalized predecessor
|
||||
before a dependent lane starts.
|
||||
|
||||
The runner writes successful checkpoint artifacts only after complete accepted
|
||||
validation. Chunk plans follow the same rule for publication. A rejection,
|
||||
invalid structured response, or incomplete validation is never reusable state;
|
||||
the current run may still hand off an otherwise valid `warn_continue` result
|
||||
according to its terminal policy. The runner carries private reuse eligibility
|
||||
through extract, merge, normalize, and generated-reference handoff. Any stage
|
||||
derived from incomplete validation skips both checkpoint lookup and all
|
||||
checkpoint publication even when that stage's own validation completes.
|
||||
External references and fully validated generated references remain eligible.
|
||||
Attempt debug records retain safe kind,
|
||||
validator, repair-usage, policy, and terminal-decision provenance. Full
|
||||
assistant and correction content remains confined to the requested detailed
|
||||
LLM trace.
|
||||
|
||||
Debug recording is attempt-scoped and application-owned. A failure to persist
|
||||
required debug data is a framework error. State roots, persistence, reason-code
|
||||
meanings, resume, and cleanup are intentionally owned by
|
||||
|
||||
@@ -47,6 +47,9 @@ The serialized
|
||||
they do not describe a current public state surface.
|
||||
|
||||
Ordered-step lane checkpoints include the step identity in their storage scope.
|
||||
Accepted step and lane identities are encoded injectively before becoming
|
||||
filesystem path components, while ordinary safe identifiers retain their
|
||||
readable paths.
|
||||
When a later lane consumes a generated artifact, its dependency fingerprints
|
||||
include the producer's artifact kind, complete schema identity, media type,
|
||||
canonical content digest, and size. Ordinary resume compares those fingerprints
|
||||
|
||||
@@ -5,6 +5,23 @@ This is the canonical guide for operating Notarius runtime state. The
|
||||
[Configuration](config.md) owns fields, defaults, and precedence. Maintainers
|
||||
who need implementation mechanics should read [Run State Internals](internal/state.md).
|
||||
|
||||
## Source Deployment
|
||||
|
||||
Linux is the supported deployment platform. Install a pinned source release
|
||||
with the Go version declared in `go.mod` (currently Go 1.25.5):
|
||||
|
||||
~~~sh
|
||||
GOWORK=off go install \
|
||||
gitea.maximumdirect.net/eric/notarius/cmd/notarius@vMAJOR.MINOR.PATCH
|
||||
~~~
|
||||
|
||||
Pin the exact tag in deployment automation rather than following a branch.
|
||||
Use [`notarius --version`](cli.md#command-summary) as a diagnostic after
|
||||
installation; its syntax and semantics are owned by the [CLI reference](cli.md).
|
||||
The maintainer publication process, including tag guards and verification,
|
||||
belongs to [Source Releases](release.md). macOS builds are best-effort for
|
||||
development, and Windows is unsupported.
|
||||
|
||||
## State Surfaces
|
||||
|
||||
Each run can use independent roots with different retention and access-control
|
||||
@@ -72,6 +89,9 @@ provider call or credentials:
|
||||
notarius config validate --config /etc/notarius/config.yml --pipeline dnd-session
|
||||
~~~
|
||||
|
||||
An unset optional `api_key_env` reaches the provider without authorization and
|
||||
may receive a 401 or 403 response.
|
||||
|
||||
Profile paths are currently resolved from the process working directory, not
|
||||
from the configuration file. The complete example's
|
||||
`./examples/profiles/dnd-extraction.yml` path is valid for a repository-root
|
||||
@@ -89,6 +109,33 @@ On success, the command reports the output bundle path. A warning-bearing run
|
||||
still succeeds and reports its warning count on standard error. Errors and
|
||||
their exit classes are defined in the [CLI reference](cli.md#output-streams-and-exit-statuses).
|
||||
|
||||
## Validation Retries And Terminal Outcomes
|
||||
|
||||
Each producer binding has one outer **retries** budget. It covers complete
|
||||
producer attempts for operational failures, invalid structured output,
|
||||
normalizer fallback retry, and semantic correction. It is independent from
|
||||
PromptKit's structural-repair calls inside one completion and from an
|
||||
LLM-backed validator's own retry budget. A semantic correction rebuilds the
|
||||
ordinary producer request and supplies only the latest rejected model response
|
||||
plus aggregated validator guidance; it is not a conversation replay.
|
||||
|
||||
After the applicable budgets are exhausted, the resolved
|
||||
[`validation_policy`](config.md#pipelines) determines the result. Structural
|
||||
failure and semantic rejection normally fail the run; an explicit
|
||||
`reject_output` records a rejection and allows unrelated work to finish. A
|
||||
validator execution failure normally uses `warn_continue`, which keeps an
|
||||
otherwise accepted result in the current run with `incomplete` validation
|
||||
provenance. It emits one bounded warning for every validator whose execution
|
||||
budget was exhausted. A corrected result that later passes validation does not
|
||||
retain abandoned-attempt warnings.
|
||||
|
||||
Treat a successful process exit as a completed run, not as proof that every
|
||||
candidate was fully validated. Inspect the receipt's `validation_status`,
|
||||
`validation_summaries`, rejection count, and warning count when an orchestrator
|
||||
requires complete validation. The durable fields and their meanings are owned
|
||||
by the [run-result receipt](integrations/run-result.md) and
|
||||
[published JSON output contract](integrations/json-output.md).
|
||||
|
||||
## Output Bundles
|
||||
|
||||
Each successful run receives a generated safe run identifier and writes beneath:
|
||||
@@ -110,9 +157,9 @@ are defined in [Accepted Chunk Map](integrations/chunk-map.md). An optional
|
||||
[evidence context](integrations/evidence-context.md) contains source-unit text
|
||||
and metadata. It is not a cache or debug artifact: retain it with the output
|
||||
bundle only for as long as consumers need it, and apply source-content access
|
||||
controls to the entire bundle. Selected lanes may collectively cite most of a
|
||||
transcript, so a broad allowlist can make the evidence artifact nearly as
|
||||
sensitive and large as the source itself.
|
||||
controls to the entire bundle. Its selected source-unit excerpt may include
|
||||
every source unit once when coverage is broad or its configured window is
|
||||
large, so do not assume a byte or token reduction or reduced sensitivity.
|
||||
|
||||
## Chunk-Plan Cache
|
||||
|
||||
@@ -138,7 +185,9 @@ The configured cache mode controls one invocation:
|
||||
A reused plan is still materialized and validated against the current source.
|
||||
If a prior plan no longer gives acceptable results, use a refresh run rather
|
||||
than editing cache files. Deleting a plan is recoverable but can repeat costly
|
||||
chunking work.
|
||||
chunking work. A plan accepted only under incomplete validation is not
|
||||
published, and a rejected cache hit falls through to ordinary generation rather
|
||||
than becoming a correction candidate.
|
||||
|
||||
## Checkpoint Recording, Resume, And Recompute
|
||||
|
||||
@@ -163,6 +212,12 @@ Reasoning-effort inheritance, replacement, and explicit clearing are distinct
|
||||
runtime identities, so checkpoints created under one state are not reused by
|
||||
either of the others.
|
||||
|
||||
Only accepted, completely validated chunk, extract, merge, and normalize
|
||||
results are checkpointed for reuse. Rejected, structurally invalid, and
|
||||
validation-incomplete producer results remain non-reusable, even when a
|
||||
`warn_continue` result advanced during its original run. A resumed invocation
|
||||
therefore reruns that producer rather than treating degraded state as accepted.
|
||||
|
||||
Checkpoint state is confined below an identity-specific path:
|
||||
|
||||
~~~
|
||||
@@ -217,13 +272,16 @@ Only a [debug-enabled run](cli.md#run) creates a bundle:
|
||||
~~~
|
||||
|
||||
The summary contains redacted invocation and resolution information plus run,
|
||||
warning, checkpoint, chunk-plan, and terminal reporting artifacts. The trace
|
||||
contains allowlisted application diagnostic records and can include source or
|
||||
derived application data. Neither surface is a cache input. Do not treat a
|
||||
debug bundle as safe to share merely because its configuration summary is
|
||||
redacted. Invocation metadata omits reasoning effort when it is inherited,
|
||||
records the replacement value when one is supplied, and records an empty value
|
||||
when inherited reasoning was explicitly cleared.
|
||||
warning, checkpoint, chunk-plan, and terminal reporting artifacts. Attempt
|
||||
terminal records contain bounded attempt kinds, validator outcomes, policy,
|
||||
decision, PromptKit repair count, and usage; they do not contain assistant
|
||||
responses or complete correction messages. The trace contains allowlisted
|
||||
application diagnostic records and can include source, model, and correction
|
||||
content. Neither surface is a cache input. Do not treat a debug bundle as safe
|
||||
to share merely because its configuration summary is redacted. Invocation
|
||||
metadata omits reasoning effort when it is inherited, records the replacement
|
||||
value when one is supplied, and records an empty value when inherited reasoning
|
||||
was explicitly cleared.
|
||||
|
||||
Notarius never creates debug state without an explicit request and never
|
||||
automatically deletes a requested bundle. If allocation succeeds, the command
|
||||
@@ -255,14 +313,29 @@ Provider execution settings and the generation timeout come from the selected
|
||||
PromptKit profile. The invocation-only **--reasoning-effort** and
|
||||
**--clear-reasoning-effort** controls may replace or clear that profile setting
|
||||
for all LLM-backed calls in one run without changing the profile. PromptKit
|
||||
v0.5.0 does not add a provider retry loop. Notarius binding retries rerun the
|
||||
complete module operation and validation chain as defined by
|
||||
[module bindings](config.md#module-bindings-and-validators).
|
||||
structural output repair happens within one structured-completion call. Its
|
||||
effective `structured_output_repair_attempts` limit is resolved from the
|
||||
selected binding, then the pipeline, then the prompt declaration; see
|
||||
[module bindings](config.md#module-bindings-and-validators). This is distinct
|
||||
from Notarius binding **retries**, which rerun the complete module operation
|
||||
and validation chain and do not consume or replenish the structural-repair
|
||||
limit. The maintained production prompts declare one repair attempt, paid only
|
||||
after a structural failure. One structured completion with repair budget **R**
|
||||
makes at most **R + 1** serial provider calls. If one stage attempt performs
|
||||
**C** structured completions, a binding with **retries: N** has a maximum of
|
||||
**(N + 1) * C * (R + 1)** provider calls; LLM-backed validators have their own
|
||||
corresponding invocation counts and budgets. This is an upper bound, not a
|
||||
promise that every call reaches a provider.
|
||||
|
||||
Timeouts are layered. Caller cancellation is the outer authority. A positive
|
||||
effective generation timeout adds an inner request deadline, while zero
|
||||
disables only that generation deadline. The HTTP client timeout remains a
|
||||
transport-wide cap. Notarius does not add another timeout around PromptKit.
|
||||
Repairs are serial within the same caller context, so their worst-case latency
|
||||
and cost follow the provider-call bound above; provision run deadlines and
|
||||
provider budgets accordingly. Credentials remain optional unless the selected
|
||||
PromptKit profile requires one, in which case preparation fails before a
|
||||
provider call when its configured credential is unavailable.
|
||||
The pinned upstream boundary and profile-format links are in
|
||||
[PromptKit Integration](integrations/pkg-promptkit.md).
|
||||
|
||||
@@ -279,6 +352,11 @@ positive value makes the effective active local-generation bound the smaller
|
||||
of **total_llm** and that local limit, so a local limit of four permits no more
|
||||
than four active local generations.
|
||||
|
||||
The Notarius scheduler admits one logical structured completion and holds that
|
||||
permit while PromptKit performs its serial corrective calls. PromptKit applies
|
||||
its selected-backend admission to each provider call; Notarius does not
|
||||
reacquire a permit or add another scheduler for a repair.
|
||||
|
||||
For a positive local limit, PromptKit owns its default waiting capacity and
|
||||
admission behavior. When a PromptKit backend has admitted all active and queued
|
||||
work, a new call fails as capacity exhaustion before generation. The adapter
|
||||
|
||||
@@ -24,6 +24,12 @@ DAGs or a general workflow language. Every stage remains explicit; general
|
||||
chunking, merging, or normalization behavior must not be hidden inside an
|
||||
extractor.
|
||||
|
||||
A stage module is one configured implementation of one pipeline stage. An
|
||||
artifact family is the cohesive domain feature that owns an artifact across
|
||||
the explicit stages and supporting codecs, validators, prompts, identity
|
||||
rules, and reference projections. Artifact-family ownership does not combine
|
||||
stages or alter the fixed pipeline.
|
||||
|
||||
Input and chunking are pipeline-wide. Each selected artifact lane owns its
|
||||
extract, merge, and normalize stages, and the output stage aggregates the run's
|
||||
lane outcomes.
|
||||
@@ -39,6 +45,12 @@ implementations. Domain-neutral model and framework layers provide reusable
|
||||
policy, contracts, and orchestration. Concrete input, pipeline, output, and
|
||||
validation extensions depend inward on those generic layers.
|
||||
|
||||
Semantic reconciliation is one such domain-neutral framework mechanism. It
|
||||
prepares bounded source context, invokes a shared model-judgment protocol,
|
||||
validates proposals, and applies safe plans through typed policies supplied by
|
||||
the consuming artifact family. It does not own domain identity, durable IDs,
|
||||
warning semantics, or artifact construction rules.
|
||||
|
||||
Generic layers must not depend on production extensions. Concrete extensions
|
||||
must not compose the application or take ownership of process behavior. The
|
||||
current packages implementing these layers are inventoried in
|
||||
@@ -155,18 +167,29 @@ starting, waits for started work, and prevents output encoding.
|
||||
## Validation
|
||||
|
||||
Validation is a framework-managed boundary around outputs from chunk, extract,
|
||||
merge, and normalize stages. Validators receive immutable stage output
|
||||
and make an explicit whole-output decision: approve, approve with warnings, or
|
||||
reject.
|
||||
merge, and normalize stages. Validators receive immutable stage output and
|
||||
make an explicit whole-output decision: approve, approve with warnings,
|
||||
reject, fail, or skip when a runtime prerequisite is unavailable.
|
||||
|
||||
Typed artifact validators receive the domain value directly. Chunk validators
|
||||
receive source-zone chunks, while serialized validators receive immutable
|
||||
representation bytes and declared schema metadata. A validator registered for
|
||||
one target or artifact kind cannot satisfy an incompatible selection.
|
||||
|
||||
Rejection is a recorded pipeline outcome, not a framework execution error.
|
||||
Validator execution failures are framework errors. Rejected output does not
|
||||
advance to the next stage.
|
||||
The framework runs every applicable validator sequentially in configured order.
|
||||
It aggregates rejections, exhausted validator failures, and skips before the
|
||||
producer policy chooses a disposition. A completed rejection never advances.
|
||||
With no rejection, an exhausted validator failure may fail the run or, under
|
||||
the configured `warn_continue` policy, advance a structurally valid candidate
|
||||
with explicit incomplete-validation provenance. Validators report findings;
|
||||
they do not choose candidate disposition.
|
||||
|
||||
A completed rejection supplies a stable reason code for internal provenance
|
||||
and bounded actionable correction guidance for the candidate producer. Reason
|
||||
codes, validator keys, and operator-facing messages remain diagnostic data;
|
||||
they are not model instructions. The framework constructs model-facing retry
|
||||
text only from the semantic guidance and fails the contract rather than
|
||||
inventing or truncating missing guidance.
|
||||
|
||||
Default validator chains are production composition policy and are registered
|
||||
centrally by stage and module. Configuration may replace a stage-local default,
|
||||
@@ -183,6 +206,30 @@ The caller of the LLM owns prompt selection, prompt inputs, response schema,
|
||||
and interpretation of structured output. Provider adapters do not own source-
|
||||
or domain-specific prompt logic.
|
||||
|
||||
PromptKit owns bounded structural correction within one structured completion.
|
||||
Notarius owns outer stage attempts, semantic validation, and acceptance policy;
|
||||
the two budgets must remain separate.
|
||||
|
||||
An LLM-backed producer can participate in semantic correction only when it
|
||||
declares `single_response_v1` and returns the exact one response that directly
|
||||
controlled its candidate. On an actionable rejection, the framework rebuilds
|
||||
the ordinary request and appends only the latest defective response as an
|
||||
`assistant` message plus one aggregated `user` correction message. This is a
|
||||
fresh replacement request, not a growing conversation. The retry budgets,
|
||||
terminal policy, and sensitive-data rationale are recorded in
|
||||
[ADR-0014](../adr/0014-feedback-aware-validation-retries.md).
|
||||
|
||||
When a model selects an application entity, callers must supply a contextual
|
||||
selection and deterministically attach the opaque application identity whenever
|
||||
the selection resolves exactly. Models do not receive or reproduce opaque
|
||||
application identifiers. Semantic reconciliation may instead expose
|
||||
contiguous, one-based candidate handles that exist only for one request;
|
||||
deterministic code resolves them before typed application, and they never
|
||||
become durable identity. This is the approved request-local-label application
|
||||
of [ADR-0012](../adr/0012-resolve-opaque-entity-identifiers-deterministically.md)
|
||||
recorded by
|
||||
[ADR-0013](../adr/0013-use-request-local-candidate-handles-for-semantic-reconciliation.md).
|
||||
|
||||
LLM calls and other external operations accept cancellation and respect
|
||||
timeouts. Concurrency control belongs in shared runtime plumbing rather than in
|
||||
individual modules.
|
||||
@@ -206,7 +253,8 @@ invalid or incompatible.
|
||||
|
||||
Run manifests record enough resolved pipeline, module, source, reference, and
|
||||
LLM provenance to make a run auditable after configuration changes. Manifests
|
||||
record identities and summaries rather than secret or large payload content.
|
||||
record identities and bounded validation summaries rather than secret, raw
|
||||
model, correction, or large payload content.
|
||||
|
||||
## State, Output, And Safety
|
||||
|
||||
@@ -226,18 +274,39 @@ an invocation that explicitly requests resume. Debug is never a cache input and
|
||||
is never created without an explicit request. Pipeline modules receive
|
||||
collaborator interfaces and never physical roots.
|
||||
|
||||
Only accepted, completely validated producer output is reusable checkpoint or
|
||||
chunk-plan state. Rejected, structurally invalid, and validation-incomplete
|
||||
results cannot become cache or checkpoint inputs, even when a
|
||||
`warn_continue` result is allowed to advance in the current run. This
|
||||
ineligibility follows derived merge and normalize results and generated
|
||||
references for the remainder of the run: current-run handoff remains allowed,
|
||||
but no dependent cache or checkpoint may be loaded or published.
|
||||
|
||||
Writes are atomic where practical. Paths for writes, moves, overwrites, and
|
||||
deletion must be narrow and explicit. Notarius never automatically deletes
|
||||
output or requested debug bundles; cache cleanup is explicit and recoverable.
|
||||
|
||||
Secrets must not appear in errors, logs, output, cache, debug summaries,
|
||||
traces, manifests, documentation, examples, or redacted configuration. Debug
|
||||
traces, manifests, documentation, examples, or redacted configuration. Raw
|
||||
assistant responses and complete correction messages are attempt-local and are
|
||||
excluded from ordinary durable records and summaries; the requested detailed
|
||||
debug trace is the sole diagnostic surface allowed to retain them. Debug
|
||||
collection is allowlisted to application-owned payloads and must not capture
|
||||
unrelated process environment values or filesystem content. Trace data may
|
||||
contain application data and therefore inherits its sensitivity; operators own
|
||||
access controls and retention. Physical layout and operation are defined in
|
||||
[Operations](../operations.md).
|
||||
|
||||
## Platform And Distribution
|
||||
|
||||
Linux is the supported deployment platform. macOS is supported only as a
|
||||
best-effort development and compilation environment, while Windows is
|
||||
unsupported. Notarius distributes source releases only: an immutable source
|
||||
tag and its checked-in release note identify a release. The project does not
|
||||
publish executable binaries, archives, installers, container images,
|
||||
checksums, signatures, or package-manager entries. Maintainer release commands
|
||||
and tag guards belong to [Source Releases](../release.md).
|
||||
|
||||
## Architectural Non-Goals
|
||||
|
||||
Notarius does not aim to provide:
|
||||
|
||||
@@ -65,6 +65,8 @@ secret values.
|
||||
| CLI contract | `docs/cli.md` | Commands, arguments, flags, invocation semantics, and exit codes. | End-to-end operating procedures, configuration field definitions, runtime filesystem layout, module implementation details. |
|
||||
| Configuration contract | `docs/config.md` | Discovery and precedence, file schema, fields, defaults, environment overrides, validation rules, and user-selectable module or validator keys. | Complete example files, CLI syntax, runtime state lifecycle, module implementation details. |
|
||||
| Operations | `docs/operations.md` | Runtime workflows, physical filesystem and state layout, output, cache, and debug handling, resume, cleanup, permissions, recovery, and operational limits. | CLI flag syntax, configuration field definitions, logical output schemas, implementation mechanics. |
|
||||
| Source release procedure | `docs/release.md` | Maintainer release selection, candidate validation, tagging, publication guards, verification, and immutable-tag recovery. | Product installation summary, CLI version semantics, historical release summaries, CI implementation detail. |
|
||||
| Release-note history | `docs/releases/` | One checked-in historical summary for each source release made under the procedure. The note at the immutable tag is that release's record. | Current commands, behavior, contracts, and compatibility definitions. |
|
||||
| Public HTTP contract, if introduced | `docs/api.md` | Routes, authentication, media types, request and response schemas, status codes, pagination, caching, idempotency, rate limits, and HTTP retry semantics. | Client walkthroughs, upstream or downstream integration internals, implementation detail. |
|
||||
| Consumer guidance, if a public package or API is introduced | `docs/consumers/` | Task-oriented use of the public interface, minimal client examples, and consumer responsibilities. | HTTP wire semantics, external protocol contracts, internal implementation detail. |
|
||||
| External and durable integration contracts | `docs/integrations/` | External file formats and protocols, upstream and downstream contracts, logical output bundle paths and schemas, media types, and compatibility behavior. | Physical runtime placement and lifecycle, internal transformations, CLI syntax, configuration defaults. |
|
||||
@@ -95,6 +97,14 @@ runtime state and how to operate or recover the application. When a workflow
|
||||
crosses these topics, choose the document that owns the task and link to the
|
||||
other contracts.
|
||||
|
||||
### Releases
|
||||
|
||||
`docs/release.md` owns the source-release procedure. Release notes are
|
||||
historical summaries, not current-state contract owners: the checked-in note at
|
||||
an immutable tag records that release, while current canonical documentation
|
||||
must change with the behavior it describes. Do not use a release note to defer
|
||||
or replace current documentation updates.
|
||||
|
||||
### Contracts And Implementation
|
||||
|
||||
Integration and API documents define externally observable shapes and
|
||||
|
||||
164
docs/release.md
Normal file
164
docs/release.md
Normal file
@@ -0,0 +1,164 @@
|
||||
# Source Releases
|
||||
|
||||
This procedure is for maintainers publishing Notarius source releases. A
|
||||
release is an immutable lightweight `vMAJOR.MINOR.PATCH` tag on `main` together
|
||||
with its checked-in `docs/releases/<tag>.md` note. Tag CI validates that source
|
||||
candidate after publication; it does not publish or repair a release.
|
||||
|
||||
Notarius publishes no binaries, archives, checksums, signatures, containers,
|
||||
package-manager entries, or Gitea release objects. Windows is not supported.
|
||||
Do not create retrospective notes for the pre-procedure `v0.1.0`, `v0.2.0`, or
|
||||
`v0.3.0` tags.
|
||||
|
||||
## Select And Describe The Release
|
||||
|
||||
Choose an unused stable semantic version in the form `vMAJOR.MINOR.PATCH`.
|
||||
Prereleases are not supported. Before `v1.0.0`, a minor release may change a
|
||||
documented CLI, configuration, durable artifact, integration, or operating
|
||||
contract when its note explains the impact and required operator action. A
|
||||
patch release must not intentionally break those documented contracts within
|
||||
its minor line.
|
||||
|
||||
Create the version-matched note as part of the candidate. Every new note uses
|
||||
this structure, with concise, truthful content in each section:
|
||||
|
||||
```markdown
|
||||
# Notarius vMAJOR.MINOR.PATCH
|
||||
|
||||
This release ...
|
||||
|
||||
## Summary
|
||||
|
||||
## Compatibility
|
||||
|
||||
## Upgrade
|
||||
|
||||
## Changes
|
||||
```
|
||||
|
||||
The note is a historical summary. Link to current canonical documentation for
|
||||
exact behavior, and update that documentation in the candidate rather than
|
||||
using the note as a substitute.
|
||||
|
||||
## Prepare The Candidate
|
||||
|
||||
Set the selected release version and disable Go workspace use for every
|
||||
candidate command:
|
||||
|
||||
```sh
|
||||
RELEASE_VERSION=vMAJOR.MINOR.PATCH
|
||||
export RELEASE_VERSION GOWORK=off
|
||||
```
|
||||
|
||||
Run the shared source-candidate checks from the repository. They cover module
|
||||
hygiene, tests, race tests, vet, builds, formatting, whitespace, maintained
|
||||
configuration validation, and the Linux and Darwin command-build matrix:
|
||||
|
||||
```sh
|
||||
./scripts/check-release-source.sh "$RELEASE_VERSION"
|
||||
```
|
||||
|
||||
Before committing, manually follow every changed local Markdown link and
|
||||
review the candidate for unintended files, generated output, credentials, or
|
||||
other unrelated changes. Commit the release note and all affected current
|
||||
documentation, then run the shared checker against that exact candidate. Push
|
||||
the candidate commit to `main` only after it succeeds. Record the exact commit
|
||||
only after that push:
|
||||
|
||||
```sh
|
||||
RELEASE_COMMIT=$(git rev-parse 'HEAD^{commit}')
|
||||
export RELEASE_COMMIT
|
||||
```
|
||||
|
||||
For private-module installation, configure standard `GOPRIVATE` matching this
|
||||
module and ordinary Git authentication for the hosting service before running
|
||||
the verification below. The exact authentication mechanism belongs to the
|
||||
maintainer environment; never record credentials or environment dumps in a
|
||||
release note, command history, or repository file.
|
||||
|
||||
## Guard And Publish The Tag
|
||||
|
||||
Fetch current remote references, then run this guard without editing the
|
||||
candidate. It requires `main`, a clean worktree and index, disabled workspace
|
||||
use, a stable release version, the recorded and pushed commit, a matching note,
|
||||
and unused local and remote tags:
|
||||
|
||||
```sh
|
||||
git fetch origin main --tags
|
||||
|
||||
if ! printf '%s\n' "$RELEASE_VERSION" |
|
||||
grep -E -x 'v(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)' >/dev/null
|
||||
then
|
||||
printf '%s\n' "invalid release version: $RELEASE_VERSION" >&2
|
||||
exit 1
|
||||
fi
|
||||
test "$GOWORK" = off
|
||||
test "$(git branch --show-current)" = main
|
||||
test -z "$(git status --porcelain)"
|
||||
test "$RELEASE_COMMIT" = "$(git rev-parse 'HEAD^{commit}')"
|
||||
test "$RELEASE_COMMIT" = "$(git rev-parse 'origin/main^{commit}')"
|
||||
test -s "docs/releases/$RELEASE_VERSION.md"
|
||||
grep -F -x "# Notarius $RELEASE_VERSION" "docs/releases/$RELEASE_VERSION.md"
|
||||
for heading in '## Summary' '## Compatibility' '## Upgrade' '## Changes'; do
|
||||
grep -F -x "$heading" "docs/releases/$RELEASE_VERSION.md"
|
||||
done
|
||||
if git rev-parse -q --verify "refs/tags/$RELEASE_VERSION" >/dev/null; then
|
||||
printf '%s\n' "local tag already exists: $RELEASE_VERSION" >&2
|
||||
exit 1
|
||||
fi
|
||||
if git ls-remote --exit-code --tags origin "refs/tags/$RELEASE_VERSION" >/dev/null 2>&1; then
|
||||
printf '%s\n' "remote tag already exists: $RELEASE_VERSION" >&2
|
||||
exit 1
|
||||
fi
|
||||
```
|
||||
|
||||
Create an explicitly lightweight tag against the guarded commit, verify its
|
||||
target, and push only that tag ref:
|
||||
|
||||
```sh
|
||||
git -c tag.gpgSign=false tag "$RELEASE_VERSION" "$RELEASE_COMMIT"
|
||||
test "$(git cat-file -t "$RELEASE_VERSION")" = commit
|
||||
test "$(git rev-parse "$RELEASE_VERSION^{commit}")" = "$RELEASE_COMMIT"
|
||||
git push origin "refs/tags/$RELEASE_VERSION:refs/tags/$RELEASE_VERSION"
|
||||
```
|
||||
|
||||
Never use `git push --tags`, move a published tag, or delete a published tag.
|
||||
|
||||
## Verify The Published Release
|
||||
|
||||
Confirm that the remote tag still points at the guarded commit and that the
|
||||
note is available from the tagged tree:
|
||||
|
||||
```sh
|
||||
REMOTE_TAG_COMMIT=$(git ls-remote origin "refs/tags/$RELEASE_VERSION" | awk '{print $1}')
|
||||
test "$REMOTE_TAG_COMMIT" = "$RELEASE_COMMIT"
|
||||
git show "$RELEASE_VERSION:docs/releases/$RELEASE_VERSION.md" >/dev/null
|
||||
```
|
||||
|
||||
Verify a fresh source installation and its diagnostic version. The temporary
|
||||
directory confines the installed command to this check:
|
||||
|
||||
```sh
|
||||
release_verification_dir=$(mktemp -d)
|
||||
trap 'rm -rf "$release_verification_dir"' 0 HUP INT TERM
|
||||
mkdir -p "$release_verification_dir/bin"
|
||||
GOWORK=off GOBIN="$release_verification_dir/bin" go install \
|
||||
"gitea.maximumdirect.net/eric/notarius/cmd/notarius@$RELEASE_VERSION"
|
||||
test "$("$release_verification_dir/bin/notarius" --version)" = "notarius $RELEASE_VERSION"
|
||||
```
|
||||
|
||||
An exact fresh checkout and `GOWORK=off go build ./cmd/notarius` is an
|
||||
equivalent source verification when local installation policy requires it.
|
||||
`notarius --version` is diagnostic only; downstream compatibility remains
|
||||
defined by the published receipt and artifact contracts.
|
||||
|
||||
## Failure And Correction Policy
|
||||
|
||||
If candidate validation fails before publication, fix the candidate on `main`,
|
||||
rerun the shared checker, and repeat the guards. An unpublished local tag may
|
||||
be deleted after inspection.
|
||||
|
||||
If the remote tag or tag CI reveals a defect, leave the published tag intact.
|
||||
Fix the defect on `main`, choose a new patch version, write a new matching
|
||||
note, and repeat this procedure. Do not weaken tag immutability or add release
|
||||
assets as a workaround.
|
||||
83
docs/releases/v0.4.0.md
Normal file
83
docs/releases/v0.4.0.md
Normal file
@@ -0,0 +1,83 @@
|
||||
# Notarius v0.4.0
|
||||
|
||||
This release strengthens LLM reliability and validation throughout the
|
||||
configured pipeline, upgrades the PromptKit integration, and establishes the
|
||||
source-release and downstream-consumer workflows needed for broader D&D
|
||||
pipeline integration.
|
||||
|
||||
## Summary
|
||||
|
||||
Notarius now distinguishes PromptKit structural-output repair from
|
||||
application-owned semantic validation retries. Producer candidates can run
|
||||
through complete deterministic validator chains, receive bounded semantic
|
||||
correction guidance, and retry under explicit stage policies. Final run
|
||||
receipts and manifests preserve bounded validation provenance, while outputs
|
||||
that advance with incomplete validation remain available to the current run
|
||||
without entering reusable checkpoint state.
|
||||
|
||||
The release also adds a maintained complete D&D subprocess-consumer workflow,
|
||||
diagnostic build versions, and the source-only release procedure used to
|
||||
publish this version.
|
||||
|
||||
## Compatibility
|
||||
|
||||
- Configuration files must use schema version 4. Version 3 is not decoded or
|
||||
rewritten; rename the top-level `scriptorium` section to `promptkit` when
|
||||
migrating. See [Configuration](../config.md#migrating-version-3-configuration).
|
||||
- PromptKit is pinned to v0.9.0. Operator profile files use PromptKit's v0.9.0
|
||||
format and may use its profile-inheritance support. Notarius continues to
|
||||
resolve operator profiles before embedded fallbacks.
|
||||
- Structural-output repair and semantic stage retries are separate bounded
|
||||
mechanisms. Maintained production prompts request one structural repair by
|
||||
default; explicit configuration can override the supported repair count.
|
||||
- Validation policy can now fail a run, reject an output, or permit an
|
||||
otherwise valid candidate to advance with incomplete-validation provenance.
|
||||
The application defaults are documented in
|
||||
[Configuration](../config.md#pipelines).
|
||||
- The `notarius.run-result.v1` receipt remains at schema version 1 and adds
|
||||
optional validation summaries plus a required validation-status field.
|
||||
Consumers of this pre-release contract should follow the current
|
||||
[run-result receipt](../integrations/run-result.md).
|
||||
- Existing D&D artifact schema identities remain unchanged. Validation and
|
||||
producer-policy changes can nevertheless cause previously accepted weak
|
||||
candidates to retry, reject, or fail instead.
|
||||
|
||||
## Upgrade
|
||||
|
||||
1. Migrate every Notarius configuration to version 4 and rename `scriptorium`
|
||||
to `promptkit`.
|
||||
2. Review deployed PromptKit profiles against the pinned v0.9.0 profile format
|
||||
and ensure their credential environment variables are available at run
|
||||
time.
|
||||
3. Run `notarius config validate --config <path> --pipeline <id>` before the
|
||||
first production invocation.
|
||||
4. Review `structured_output_repair_attempts`, producer retry counts, and
|
||||
`validation_policy` wherever the deployment needs behavior different from
|
||||
the documented defaults.
|
||||
5. Update subprocess consumers to inspect receipt `validation_status` and to
|
||||
tolerate the optional bounded `validation_summaries` field. A consumer that
|
||||
requires fully validated artifacts should require `approved`.
|
||||
|
||||
## Changes
|
||||
|
||||
- Upgraded PromptKit from v0.5.0 through v0.9.0 and adopted profile
|
||||
inheritance, structured-output repair, typed error classification, and the
|
||||
correction-aware completion protocol.
|
||||
- Added pipeline and binding configuration for structural repair and terminal
|
||||
validation policy, with strict startup validation and effective-setting
|
||||
provenance.
|
||||
- Added feedback-aware retries for chunking, extraction, merge, normalize, and
|
||||
semantic reconciliation producers. Retry prompts contain the exact defective
|
||||
response and actionable semantic correction guidance without exposing
|
||||
internal reason codes or opaque entity identifiers.
|
||||
- Added complete validator-chain execution, validator retry handling, bounded
|
||||
warnings, terminal dispositions, and durable validation summaries.
|
||||
- Prevented validation-incomplete artifacts and all derived lineage from
|
||||
loading or publishing reusable checkpoints while preserving same-run
|
||||
generated-reference handoff.
|
||||
- Tightened cached chunk-plan validation so only completely validated plans are
|
||||
reused or replace stored plans.
|
||||
- Added a machine-readable subprocess receipt workflow and complete D&D
|
||||
consumer documentation covering all maintained artifacts.
|
||||
- Added source-release checks, immutable lightweight-tag guidance, Linux and
|
||||
Darwin build verification, and diagnostic `notarius --version` output.
|
||||
2474
docs/roadmap/archive/audit.md
Normal file
2474
docs/roadmap/archive/audit.md
Normal file
File diff suppressed because it is too large
Load Diff
@@ -5,13 +5,90 @@ configuration, operations, internal, and integration docs. This roadmap records
|
||||
future work only. Items are ordered roughly by current value and specificity,
|
||||
not as committed release dates.
|
||||
|
||||
## Near-Term Validation And LLM Reliability
|
||||
|
||||
PromptKit now owns structural output repair within one completion. Notarius
|
||||
owns stage candidates, validator chains, semantic rejection policy, bounded
|
||||
feedback-aware stage retries, validation provenance, and reusable-state
|
||||
eligibility. The remaining near-term work applies those completed foundations
|
||||
to domain review and operator-facing diagnostics.
|
||||
|
||||
### D&D Combat Scene Semantic Validation
|
||||
|
||||
- Add an optional production LLM-backed D&D validator that determines whether
|
||||
proposed scene boundaries and classifications represent substantive active
|
||||
combat correctly. Its central quality goal is that active combat is kept in
|
||||
coherent scenes classified as `combat`, rather than split incorrectly or
|
||||
hidden inside scenes classified as `narrative`, `recap`, or `meta`.
|
||||
- Resolve the validator's exact target before implementation. The current
|
||||
`dnd/scenes` chunker owns only complete, gap-free source ranges, while the
|
||||
per-chunk `dnd/scene-descriptions` extractor owns the `combat`, `narrative`,
|
||||
`recap`, and `meta` classification. The preferred initial placement is
|
||||
therefore an extract-stage validator for `dnd/scene-descriptions`, where it
|
||||
can compare one proposed kind with the corresponding transcript chunk.
|
||||
- Consider a chunk-stage LLM validator only for a distinct boundary-coherence
|
||||
question that can be answered from the complete transcript and proposed
|
||||
range map, such as whether one continuous combat was fragmented across
|
||||
inappropriate scene boundaries. Do not duplicate the same classification
|
||||
judgment at both stages. Moving classification into chunk-plan annotations
|
||||
would change the deliberately minimal, annotation-free chunk contract and
|
||||
requires an explicit architecture review before it is selected.
|
||||
- Validate both false negatives and false positives: a non-combat kind must not
|
||||
omit substantive active combat, and a combat kind must be supported by such
|
||||
combat. Keep the existing deterministic downstream rule that combat-turn
|
||||
extraction runs only for an exact `combat` scene classification; semantic
|
||||
review improves the upstream classification but does not replace that gate.
|
||||
- Run the semantic validator through PromptKit, use a minimal required-field
|
||||
structured response schema, and let PromptKit repair structural validator
|
||||
output within its bounded budget. A contract-invalid final validator response
|
||||
is a validator execution failure, not a semantic rejection and not a reason
|
||||
to recursively validate the validator.
|
||||
- Evaluate the prompt and decision policy against a small human-reviewed set
|
||||
containing combat setup, active turns, interruptions, multi-phase encounters,
|
||||
brief rules discussion, aftermath, recalled combat, and false-positive
|
||||
hostile dialogue. Measure false acceptance, false rejection, retry success,
|
||||
added calls, latency, and token cost before placing it in the production
|
||||
default chain.
|
||||
- An ADR is not required if classification remains owned by
|
||||
`dnd/scene-descriptions` and the validator follows the generic validation ADR.
|
||||
Create or supersede an ADR if the work transfers scene classification into
|
||||
the chunker or otherwise changes stage ownership or the durable chunk-plan
|
||||
contract.
|
||||
|
||||
### Warning Signal And Presentation Reform
|
||||
|
||||
- Audit every warning producer and representative successful runs. Ordinary
|
||||
success producing dozens of warnings is a failed operator experience: the
|
||||
volume obscures actionable problems and trains operators to ignore the
|
||||
warning channel.
|
||||
- Define a small warning taxonomy that distinguishes actionable degradation,
|
||||
incomplete validation, lossy fallback, and data-quality risk from routine
|
||||
normalization observations or informational diagnostics. Preserve detailed
|
||||
traceability in debug or manifest data without promoting every observation
|
||||
to a top-level CLI warning.
|
||||
- Consider stable deduplication and aggregation by scope and reason code,
|
||||
bounded samples plus omitted counts, and a concise CLI summary with a path to
|
||||
detailed diagnostics. Do not suppress genuine validator execution failures
|
||||
merely to reduce the count.
|
||||
- Decide which warnings affect process status, rejection summaries, durable run
|
||||
receipts, or only debug output. Ensure warning ordering and aggregation are
|
||||
deterministic across concurrent execution.
|
||||
- Establish a representative warning-volume acceptance target and human review
|
||||
workflow before changing individual producers piecemeal. The intended result
|
||||
is not zero warnings; it is a small set in which every surfaced warning merits
|
||||
operator attention.
|
||||
- This work does not require an ADR unless it changes validation acceptance,
|
||||
failure, or durable contract semantics. CLI presentation and diagnostic
|
||||
taxonomy otherwise belong in a feature roadmap followed by updates to their
|
||||
canonical configuration, operations, integration, and internal documents.
|
||||
|
||||
## Near-Term D&D Pipeline
|
||||
|
||||
### Evaluate Spell Extraction And Normalization
|
||||
|
||||
- Evaluate ordinary extraction retries and the completed normalization path
|
||||
against a human-reviewed transcript set before adding repair-aware retries or
|
||||
an LLM-backed semantic validator.
|
||||
against a human-reviewed transcript set before and after adopting the shared
|
||||
PromptKit repair and Notarius validation-retry policies above.
|
||||
- Maintain a small set of human-reviewed transcripts and outputs for prompt,
|
||||
validator, and normalizer development. Treat model-quality review as an
|
||||
iterative human evaluation aid, not a deterministic correctness gate.
|
||||
@@ -24,26 +101,51 @@ not as committed release dates.
|
||||
|
||||
## Shared Normalization And Quality Work
|
||||
|
||||
### Generic LLM-Assisted Deduplication
|
||||
The implemented source-backed core and initial D&D registry adoption are
|
||||
described by [Module Internals](../internal/modules.md#semantic-reconciliation)
|
||||
and
|
||||
[D&D Module Internals](../internal/dnd.md#semantic-registry-reconciliation).
|
||||
The sections below keep broader extensions deferred.
|
||||
|
||||
- Add a reusable normalizer that asks an LLM to identify duplicate sets in a
|
||||
list and propose one replacement element for each set.
|
||||
- Define the minimum domain-neutral input contract, initially an ordered list
|
||||
whose elements have stable unique IDs. Artifact-kind registrations or
|
||||
adapters may expose that structure without moving domain rules into the
|
||||
generic package.
|
||||
- Keep mutation deterministic: parse and validate the model's duplicate groups,
|
||||
require every referenced ID to exist, reject overlapping or malformed groups,
|
||||
prevent unrelated insertion or deletion, and apply only approved replacement
|
||||
operations in code.
|
||||
- Preserve provenance needed for audit and downstream validation, and emit
|
||||
warnings describing every collapsed group.
|
||||
- Evaluate batching and context-window limits before applying the normalizer to
|
||||
large artifact collections.
|
||||
### Large-Collection Semantic Reconciliation
|
||||
|
||||
The model may use its own domain knowledge to judge semantic duplication; the
|
||||
generic implementation is responsible only for the common proposal contract,
|
||||
safety checks, and deterministic application of accepted changes.
|
||||
- Evaluate deterministic candidate blocking only after representative registry
|
||||
inputs exceed the active roadmap's bounded single-request limits. Blocking
|
||||
should use cheap, explainable signals to form plausible comparison sets while
|
||||
preserving the possibility that a duplicate appears outside a lexical name
|
||||
match.
|
||||
- Define correctness for candidates that appear in more than one block,
|
||||
conflicting canonical selections, transitive identity across blocks, retry
|
||||
isolation, and deterministic final ordering before implementation.
|
||||
- Prefer a reconciliation graph or union plan with explicit conflict checks
|
||||
over arbitrary fixed-size slices. Never silently treat a batch boundary as
|
||||
evidence that two candidates are distinct.
|
||||
- Record per-request bounds, block provenance, model calls, discarded
|
||||
proposals, and final group derivation well enough to audit a collapse.
|
||||
|
||||
### Operator-Selected Semantic Policies
|
||||
|
||||
- Consider allowing an operator to select an approved semantic-policy prompt
|
||||
for a typed reconciliation module without replacing the shared protocol,
|
||||
response schema, or deterministic safety rules.
|
||||
- Define the trusted asset source, configuration syntax, compatibility checks,
|
||||
startup validation, provenance, prompt fingerprinting, checkpoint effects,
|
||||
and support boundary before exposing the option.
|
||||
- Prefer selection among registered, typed-policy-compatible prompt assets over
|
||||
arbitrary filesystem prompt paths. Do not add this flexibility until an
|
||||
operator workflow requires it; artifact-family-owned policy remains simpler
|
||||
and safer for the initial implementation.
|
||||
|
||||
### Broader Reconciliation Inputs And Module Selection
|
||||
|
||||
- Revisit alternate context providers when a concrete non-source-backed entity
|
||||
collection needs semantic reconciliation. Any extension must preserve the
|
||||
same request-local identity, deterministic proposal validation, provenance,
|
||||
and typed application guarantees.
|
||||
- Consider a selectable generic normalizer only if Notarius gains a real
|
||||
domain-neutral typed artifact contract that can safely support it. Do not
|
||||
weaken exact artifact registration or introduce reflection-based arbitrary
|
||||
JSON mutation merely to expose a universal module key.
|
||||
|
||||
### Validation And Review
|
||||
|
||||
@@ -100,6 +202,18 @@ checkpoint reuse, when an older artifact may be decoded or adapted, and when a
|
||||
producer or all dependents must be recomputed. Do not add a general migration
|
||||
framework until an actual contract change requires one.
|
||||
|
||||
### Artifact-family-oriented physical packaging
|
||||
|
||||
[ADR-0004](../adr/0004-package-modules-by-domain.md) currently groups production
|
||||
extensions by domain and then by pipeline stage. After artifact-family
|
||||
ownership terminology is established and more families span extraction,
|
||||
normalization, validation, codecs, references, and assets, reassess whether a
|
||||
feature-first physical layout would improve navigation and reduce scattered
|
||||
changes enough to justify a repository-wide package migration. Any change must
|
||||
address Go dependency cycles, registrar ownership, stable public module keys,
|
||||
and supersession of the affected ADR-0004 decision. Conceptual artifact-family
|
||||
ownership does not by itself require this move.
|
||||
|
||||
## Blue-Sky Platform And Operations
|
||||
|
||||
These ideas are intentionally less specified. Promote one into an earlier
|
||||
@@ -115,7 +229,6 @@ section only after a concrete workflow, contract, and priority emerge.
|
||||
### Distribution And Operations
|
||||
|
||||
- Packaged release artifacts for alpha distribution.
|
||||
- A documented versioning and release process.
|
||||
- Optional generated example-output fixtures with a regeneration procedure.
|
||||
- Additional diagnostics or reporting views.
|
||||
|
||||
|
||||
7
go.mod
7
go.mod
@@ -3,9 +3,14 @@ module gitea.maximumdirect.net/eric/notarius
|
||||
go 1.25.5
|
||||
|
||||
require (
|
||||
gitea.maximumdirect.net/eric/promptkit v0.5.0
|
||||
gitea.maximumdirect.net/eric/promptkit v0.9.0
|
||||
github.com/santhosh-tekuri/jsonschema/v6 v6.0.2
|
||||
gopkg.in/yaml.v3 v3.0.1
|
||||
)
|
||||
|
||||
require golang.org/x/text v0.40.0
|
||||
|
||||
require (
|
||||
gitea.maximumdirect.net/eric/promptkit-backend-openrouter v1.0.0 // indirect
|
||||
gitea.maximumdirect.net/eric/promptkit-backend-rakestrawhome v1.0.0 // indirect
|
||||
)
|
||||
|
||||
8
go.sum
8
go.sum
@@ -1,5 +1,9 @@
|
||||
gitea.maximumdirect.net/eric/promptkit v0.5.0 h1:jnpazLyyNhWrB2xzwwtUkNUfktkTdkENTwuSPnKiYrc=
|
||||
gitea.maximumdirect.net/eric/promptkit v0.5.0/go.mod h1:R95NM6fbMDGDC0/UomgnSBP6ui2ns+8SZb8bESNvrDQ=
|
||||
gitea.maximumdirect.net/eric/promptkit v0.9.0 h1:IpvDRC8L6xRxQ9hpuyKOmMc5b6MeLTKYyx+h1YAjy08=
|
||||
gitea.maximumdirect.net/eric/promptkit v0.9.0/go.mod h1:oMJ/WUJImUtwJ5e+6MAGECPYAErAkOaKel0G+3T/b4E=
|
||||
gitea.maximumdirect.net/eric/promptkit-backend-openrouter v1.0.0 h1:lc062euk2qseO//D762i3JaFyulDNML3eQQX7DkYTho=
|
||||
gitea.maximumdirect.net/eric/promptkit-backend-openrouter v1.0.0/go.mod h1:AIa7kAu2mfrRQgcspe4L+DW51WqgnALQT60lqkEywJI=
|
||||
gitea.maximumdirect.net/eric/promptkit-backend-rakestrawhome v1.0.0 h1:j9YY7wsTVjzke2kHH4YAzpU0oUpM+x+nXwl1IeS+2eg=
|
||||
gitea.maximumdirect.net/eric/promptkit-backend-rakestrawhome v1.0.0/go.mod h1:4RNS+LILDg4JbS4Ts9Lwy1C92wauXJIbeQaalps4Koo=
|
||||
github.com/dlclark/regexp2 v1.11.0 h1:G/nrcoOa7ZXlpoa/91N3X7mM3r8eIlMBBJZvsz/mxKI=
|
||||
github.com/dlclark/regexp2 v1.11.0/go.mod h1:DHkYz0B9wPfa6wondMfaivmHpzrQ3v9q8cnmRbL6yW8=
|
||||
github.com/santhosh-tekuri/jsonschema/v6 v6.0.2 h1:KRzFb2m7YtdldCEkzs6KqmJw4nqEVZGK7IN2kJkjTuQ=
|
||||
|
||||
36
internal/buildinfo/buildinfo.go
Normal file
36
internal/buildinfo/buildinfo.go
Normal file
@@ -0,0 +1,36 @@
|
||||
// Package buildinfo resolves the product version embedded in a Notarius build.
|
||||
package buildinfo
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"regexp"
|
||||
"runtime/debug"
|
||||
)
|
||||
|
||||
var stableVersion = regexp.MustCompile(`^v(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)$`)
|
||||
|
||||
// Override is set at link time for controlled builds.
|
||||
var Override string
|
||||
|
||||
// Version returns the release version embedded in the build, or development
|
||||
// when the build does not carry a stable release tag.
|
||||
func Version() (string, error) {
|
||||
buildVersion := ""
|
||||
if info, ok := debug.ReadBuildInfo(); ok {
|
||||
buildVersion = info.Main.Version
|
||||
}
|
||||
return resolve(Override, buildVersion)
|
||||
}
|
||||
|
||||
func resolve(override, buildVersion string) (string, error) {
|
||||
if override != "" {
|
||||
if !stableVersion.MatchString(override) {
|
||||
return "", fmt.Errorf("build version override is not a stable release tag")
|
||||
}
|
||||
return override, nil
|
||||
}
|
||||
if stableVersion.MatchString(buildVersion) {
|
||||
return buildVersion, nil
|
||||
}
|
||||
return "development", nil
|
||||
}
|
||||
45
internal/buildinfo/buildinfo_test.go
Normal file
45
internal/buildinfo/buildinfo_test.go
Normal file
@@ -0,0 +1,45 @@
|
||||
package buildinfo
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestResolve(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
override string
|
||||
buildVersion string
|
||||
want string
|
||||
wantErr bool
|
||||
}{
|
||||
{name: "stable main module version", buildVersion: "v1.2.3", want: "v1.2.3"},
|
||||
{name: "zero version", buildVersion: "v0.0.0", want: "v0.0.0"},
|
||||
{name: "override takes precedence", override: "v2.3.4", buildVersion: "v1.2.3", want: "v2.3.4"},
|
||||
{name: "invalid override", override: "version", buildVersion: "v1.2.3", wantErr: true},
|
||||
{name: "override with whitespace", override: " v1.2.3", wantErr: true},
|
||||
{name: "leading zero major", buildVersion: "v01.2.3", want: "development"},
|
||||
{name: "leading zero minor", buildVersion: "v1.02.3", want: "development"},
|
||||
{name: "leading zero patch", buildVersion: "v1.2.03", want: "development"},
|
||||
{name: "build version with whitespace", buildVersion: "v1.2.3 ", want: "development"},
|
||||
{name: "prerelease", buildVersion: "v1.2.3-rc.1", want: "development"},
|
||||
{name: "build suffix", buildVersion: "v1.2.3+build.1", want: "development"},
|
||||
{name: "pseudo version", buildVersion: "v0.0.0-20260102030405-abcdef123456", want: "development"},
|
||||
{name: "development build", buildVersion: "(devel)", want: "development"},
|
||||
{name: "missing build information", want: "development"},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got, err := resolve(tt.override, tt.buildVersion)
|
||||
if tt.wantErr {
|
||||
if err == nil {
|
||||
t.Fatal("resolve() error = nil, want error")
|
||||
}
|
||||
return
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatalf("resolve() error = %v", err)
|
||||
}
|
||||
if got != tt.want {
|
||||
t.Fatalf("resolve() = %q, want %q", got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
81
internal/cli/assembled_enemy_event_codec_contract_test.go
Normal file
81
internal/cli/assembled_enemy_event_codec_contract_test.go
Normal file
@@ -0,0 +1,81 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
)
|
||||
|
||||
const invalidEnemyEventExtractorKey = "test/dnd/invalid-enemy-events"
|
||||
|
||||
func TestAssembledEnemyEventLaneRejectsInvalidFinalArtifactDespiteValidatorOverrides(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
if err := pipeline.RegisterExtractor[dnd.EnemyEventList](components.registries.Extractors, pipeline.ModuleSpec{
|
||||
Key: invalidEnemyEventExtractorKey,
|
||||
Stage: pipeline.StageExtract,
|
||||
ExecutionClass: contracts.ExecutionClassDeterministic,
|
||||
Requires: []string{"chunks", "source.transcript"},
|
||||
Provides: []string{"dnd.enemy_events"},
|
||||
ArtifactKind: dnd.EnemyEventListKind,
|
||||
}, func() (contracts.Extractor[dnd.EnemyEventList], error) {
|
||||
return invalidEnemyEventExtractor{}, nil
|
||||
}); err != nil {
|
||||
t.Fatalf("register extractor: %v", err)
|
||||
}
|
||||
|
||||
accept := pipeline.ValidatorOverride{Set: true, Validators: []pipeline.ModuleBinding{pipeline.Binding("generic/always_accept")}}
|
||||
resolved, err := pipeline.ResolvePipeline(pipeline.PipelineProfile{
|
||||
ID: "assembled-invalid-enemy-events",
|
||||
Input: pipeline.Binding("seriatim"),
|
||||
Chunk: pipeline.ModuleBinding{Module: "generic", Options: map[string]any{"max_units": 1}},
|
||||
Artifacts: map[string]pipeline.ArtifactLaneProfile{
|
||||
"enemy-events": {
|
||||
Extract: pipeline.ModuleBinding{Module: invalidEnemyEventExtractorKey, Validators: accept},
|
||||
Normalize: pipeline.ModuleBinding{Module: pipeline.DefaultNormalizeModule, Validators: accept},
|
||||
},
|
||||
},
|
||||
Output: pipeline.Binding("json"),
|
||||
}, pipeline.ResolveOptions{}, catalogFromRegistries(components.registries))
|
||||
if err != nil {
|
||||
t.Fatalf("ResolvePipeline() error = %v", err)
|
||||
}
|
||||
|
||||
prepared, err := pipeline.Prepare(resolved, components.registries, pipeline.ModuleDependencies{})
|
||||
if err != nil {
|
||||
t.Fatalf("Prepare() error = %v", err)
|
||||
}
|
||||
_, err = pipeline.New().Run(context.Background(), pipeline.RunInput{
|
||||
Prepared: prepared,
|
||||
RawInput: readRepositoryFile(t, "examples", "seriatim-minimal-transcript.json"),
|
||||
ChunkCacheMode: pipeline.ChunkCacheBypass,
|
||||
})
|
||||
if err == nil || !strings.Contains(err.Error(), "serialize accepted extract output") || !strings.Contains(err.Error(), "must not exceed") {
|
||||
t.Fatalf("Run() error = %v, want final durable range rejection", err)
|
||||
}
|
||||
}
|
||||
|
||||
type invalidEnemyEventExtractor struct{}
|
||||
|
||||
func (invalidEnemyEventExtractor) Key() string { return invalidEnemyEventExtractorKey }
|
||||
|
||||
func (invalidEnemyEventExtractor) ReferenceSlots() []contracts.ReferenceSlot { return nil }
|
||||
|
||||
func (invalidEnemyEventExtractor) Extract(ctx context.Context, req contracts.TypedExtractionRequest) (contracts.TypedExtractionResult[dnd.EnemyEventList], error) {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return contracts.TypedExtractionResult[dnd.EnemyEventList]{}, err
|
||||
}
|
||||
if req.Source == nil {
|
||||
return contracts.TypedExtractionResult[dnd.EnemyEventList]{}, errors.New("assembled extractor requires source")
|
||||
}
|
||||
return contracts.TypedExtractionResult[dnd.EnemyEventList]{Value: dnd.EnemyEventList{Events: []dnd.EnemyEvent{{
|
||||
Name: "Ashfang",
|
||||
Kind: dnd.EnemyEventKindEngaged,
|
||||
SourceRefs: []source.SourceRef{{SourceID: req.Source.ID, StartUnitID: 2, EndUnitID: 1}},
|
||||
}}}}, nil
|
||||
}
|
||||
@@ -21,6 +21,10 @@ import (
|
||||
|
||||
const assembledSpellExtractorKey = "test/dnd/spell-casts"
|
||||
|
||||
const assembledCorrectingSpellExtractorKey = "test/dnd/correcting-spell-casts"
|
||||
|
||||
const assembledDirectSpellValidatorKey = "test/dnd/direct-spell-correction"
|
||||
|
||||
func TestAssembledSpellPipelineNormalizesMergedCasts(t *testing.T) {
|
||||
registries, resolved, extractor := assembledSpellPipeline(t, assembledSpellPipelineOptions{})
|
||||
prepared, err := pipeline.Prepare(resolved, registries, pipeline.ModuleDependencies{})
|
||||
@@ -101,6 +105,30 @@ func TestAssembledSpellPipelineNormalizesMergedCasts(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestAssembledSpellPipelineCorrectsRejectedDirectExtraction(t *testing.T) {
|
||||
registries, resolved, extractor := assembledCorrectingSpellPipeline(t)
|
||||
prepared, err := pipeline.Prepare(resolved, registries, pipeline.ModuleDependencies{})
|
||||
if err != nil {
|
||||
t.Fatalf("Prepare() error = %v, want nil", err)
|
||||
}
|
||||
|
||||
output, err := pipeline.New().Run(context.Background(), pipeline.RunInput{
|
||||
Prepared: prepared,
|
||||
RawInput: readRepositoryFile(t, "examples", "seriatim-minimal-transcript.json"),
|
||||
ChunkCacheMode: pipeline.ChunkCacheBypass,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("Run() error = %v, want nil", err)
|
||||
}
|
||||
if len(output.Rejected) != 0 || output.Manifest.ValidationStatus != "approved" || len(output.NormalizeOutputs) != 1 {
|
||||
t.Fatalf("run output = %#v, want corrected accepted spell output", output)
|
||||
}
|
||||
correction := extractor.correctionSnapshot()
|
||||
if correction == nil || string(correction.AssistantResponse) != `{"spell":"Mysterious Burst"}` || !strings.Contains(correction.UserGuidance, "use a known spell name") || !strings.Contains(correction.UserGuidance, "complete corrected replacement") || strings.Contains(correction.UserGuidance, "unknown_spell") || strings.Contains(correction.UserGuidance, "spell is not in the catalog") {
|
||||
t.Fatalf("extract correction = %#v, want exact rejected model response and semantic replacement guidance only", correction)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAssembledSpellPipelineHonorsNormalizeValidatorOverride(t *testing.T) {
|
||||
registries, resolved, _ := assembledSpellPipeline(t, assembledSpellPipelineOptions{normalizeValidatorOverride: true})
|
||||
var normalizeChain *pipeline.ResolvedValidatorChain
|
||||
@@ -134,8 +162,9 @@ func TestAssembledSpellPipelineHonorsNormalizeValidatorOverride(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestAssembledSpellPipelineRejectsUnknownSpellWithoutPromotingAttemptWarning(t *testing.T) {
|
||||
func TestAssembledSpellPipelinePromotesTerminalUnknownSpellWarning(t *testing.T) {
|
||||
registries, resolved, _ := assembledSpellPipeline(t, assembledSpellPipelineOptions{unknownSpell: true})
|
||||
resolved.Steps[0].ArtifactLanes[0].NormalizeValidationPolicy.SemanticRejection = pipeline.SemanticRejectionRejectOutput
|
||||
prepared, err := pipeline.Prepare(resolved, registries, pipeline.ModuleDependencies{})
|
||||
if err != nil {
|
||||
t.Fatalf("Prepare() error = %v, want nil", err)
|
||||
@@ -161,10 +190,8 @@ func TestAssembledSpellPipelineRejectsUnknownSpellWithoutPromotingAttemptWarning
|
||||
if !reflect.DeepEqual(rejectedFile.Rejected, output.Rejected) {
|
||||
t.Fatalf("rejected file = %#v, run rejections = %#v, want durable rejection diagnostic", rejectedFile.Rejected, output.Rejected)
|
||||
}
|
||||
for _, warning := range output.Warnings {
|
||||
if warning.ReasonCode == spellnormalize.ReasonCodeSpellNameUnresolved {
|
||||
t.Fatalf("warnings = %#v, want rejected-attempt warning to remain non-durable", output.Warnings)
|
||||
}
|
||||
if len(output.Warnings) != 2 || output.Warnings[0].ReasonCode != spellnormalize.ReasonCodeSpellNameUnresolved || output.Warnings[0].Scope != "spell_casts[0]" || output.Warnings[1].ReasonCode != "spell_not_near_source" {
|
||||
t.Fatalf("warnings = %#v, want complete terminal normalize validation warnings", output.Warnings)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -208,6 +235,47 @@ type assembledSpellPipelineOptions struct {
|
||||
unknownSpell bool
|
||||
}
|
||||
|
||||
func assembledCorrectingSpellPipeline(t *testing.T) (pipeline.Registries, pipeline.ResolvedPipeline, *assembledCorrectingSpellExtractor) {
|
||||
t.Helper()
|
||||
components := productionTestComponents(t)
|
||||
extractor := &assembledCorrectingSpellExtractor{}
|
||||
if err := pipeline.RegisterExtractor[dnd.SpellList](components.registries.Extractors, pipeline.ModuleSpec{
|
||||
Key: assembledCorrectingSpellExtractorKey,
|
||||
Stage: pipeline.StageExtract,
|
||||
ExecutionClass: contracts.ExecutionClassLLMBacked,
|
||||
CorrectionProtocol: contracts.CorrectionProtocolSingleResponseV1,
|
||||
Requires: []string{"chunks", "source.transcript"},
|
||||
Provides: []string{"dnd.spell_casts"},
|
||||
ArtifactKind: dnd.SpellListKind,
|
||||
}, func() (contracts.Extractor[dnd.SpellList], error) {
|
||||
return extractor, nil
|
||||
}); err != nil {
|
||||
t.Fatalf("register correcting extractor: %v", err)
|
||||
}
|
||||
if err := pipeline.RegisterTypedValidator[dnd.SpellList](components.registries.Validators, dnd.SpellListKind, pipeline.ValidatorSpec{Key: assembledDirectSpellValidatorKey, ExecutionClass: contracts.ExecutionClassDeterministic}, func() (contracts.TypedValidator[dnd.SpellList], error) {
|
||||
return assembledDirectSpellValidator{}, nil
|
||||
}); err != nil {
|
||||
t.Fatalf("register direct spell validator: %v", err)
|
||||
}
|
||||
|
||||
extract := pipeline.Binding(assembledCorrectingSpellExtractorKey)
|
||||
extract.Retries = 1
|
||||
extract.Validators = pipeline.ValidatorOverride{Set: true, Validators: []pipeline.ModuleBinding{{Module: assembledDirectSpellValidatorKey}}}
|
||||
resolved, err := pipeline.ResolvePipeline(pipeline.PipelineProfile{
|
||||
ID: "assembled-dnd-correcting-spells",
|
||||
Input: pipeline.Binding("seriatim"),
|
||||
Chunk: pipeline.ModuleBinding{Module: "generic", Options: map[string]any{"max_units": 1}},
|
||||
Artifacts: map[string]pipeline.ArtifactLaneProfile{
|
||||
"spells": {Extract: extract, Normalize: pipeline.Binding(spellnormalize.Key)},
|
||||
},
|
||||
Output: pipeline.Binding("json"),
|
||||
}, pipeline.ResolveOptions{}, catalogFromRegistries(components.registries))
|
||||
if err != nil {
|
||||
t.Fatalf("ResolvePipeline() error = %v, want nil", err)
|
||||
}
|
||||
return components.registries, resolved, extractor
|
||||
}
|
||||
|
||||
func assembledSpellPipeline(t *testing.T, options assembledSpellPipelineOptions) (pipeline.Registries, pipeline.ResolvedPipeline, *assembledSpellExtractor) {
|
||||
t.Helper()
|
||||
components := productionTestComponents(t)
|
||||
@@ -253,6 +321,69 @@ type assembledSpellExtractor struct {
|
||||
unknownSpell bool
|
||||
}
|
||||
|
||||
type assembledCorrectingSpellExtractor struct {
|
||||
mu sync.Mutex
|
||||
correction *contracts.SemanticCorrection
|
||||
}
|
||||
|
||||
func (*assembledCorrectingSpellExtractor) Key() string { return assembledCorrectingSpellExtractorKey }
|
||||
|
||||
func (*assembledCorrectingSpellExtractor) ReferenceSlots() []contracts.ReferenceSlot { return nil }
|
||||
|
||||
func (e *assembledCorrectingSpellExtractor) Extract(_ context.Context, req contracts.TypedExtractionRequest) (contracts.TypedExtractionResult[dnd.SpellList], error) {
|
||||
if req.Source == nil || req.Chunk == nil {
|
||||
return contracts.TypedExtractionResult[dnd.SpellList]{}, fmt.Errorf("correcting assembled extractor requires source and chunk")
|
||||
}
|
||||
response := `{"spell":"accepted"}`
|
||||
value := dnd.SpellList{SpellCasts: []dnd.SpellCast{}}
|
||||
if req.Chunk.Index == 0 && req.Correction == nil {
|
||||
response = `{"spell":"Mysterious Burst"}`
|
||||
value.SpellCasts = []dnd.SpellCast{{Caster: "Aria", Spell: "Mysterious Burst", SourceRefs: []source.SourceRef{{SourceID: req.Source.ID, StartUnitID: 1, EndUnitID: 1}}}}
|
||||
}
|
||||
if req.Chunk.Index == 0 && req.Correction != nil {
|
||||
correction, err := contracts.CloneSemanticCorrection(req.Correction)
|
||||
if err != nil {
|
||||
return contracts.TypedExtractionResult[dnd.SpellList]{}, err
|
||||
}
|
||||
e.mu.Lock()
|
||||
e.correction = correction
|
||||
e.mu.Unlock()
|
||||
value.SpellCasts = []dnd.SpellCast{{Caster: "Aria", Spell: "Cure Wounds", SourceRefs: []source.SourceRef{{SourceID: req.Source.ID, StartUnitID: 1, EndUnitID: 1}}}}
|
||||
}
|
||||
candidate, err := contracts.NewModelCandidate([]byte(response), contracts.CorrectionProtocolSingleResponseV1)
|
||||
if err != nil {
|
||||
return contracts.TypedExtractionResult[dnd.SpellList]{}, err
|
||||
}
|
||||
return contracts.TypedExtractionResult[dnd.SpellList]{Value: value, ModelCandidate: candidate}, nil
|
||||
}
|
||||
|
||||
func (e *assembledCorrectingSpellExtractor) correctionSnapshot() *contracts.SemanticCorrection {
|
||||
e.mu.Lock()
|
||||
defer e.mu.Unlock()
|
||||
correction, err := contracts.CloneSemanticCorrection(e.correction)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
return correction
|
||||
}
|
||||
|
||||
type assembledDirectSpellValidator struct{}
|
||||
|
||||
func (assembledDirectSpellValidator) Name() string { return assembledDirectSpellValidatorKey }
|
||||
|
||||
func (assembledDirectSpellValidator) ExecutionClass() contracts.ExecutionClass {
|
||||
return contracts.ExecutionClassDeterministic
|
||||
}
|
||||
|
||||
func (assembledDirectSpellValidator) Validate(_ context.Context, req contracts.TypedValidationRequest[dnd.SpellList]) (contracts.ValidationResult, error) {
|
||||
for _, cast := range req.Value.SpellCasts {
|
||||
if cast.Spell == "Mysterious Burst" {
|
||||
return contracts.ValidationResult{Approved: false, ReasonCode: "unknown_spell", Message: "spell is not in the catalog", CorrectionGuidance: "use a known spell name"}, nil
|
||||
}
|
||||
}
|
||||
return contracts.ValidationResult{Approved: true}, nil
|
||||
}
|
||||
|
||||
func (e *assembledSpellExtractor) Key() string { return assembledSpellExtractorKey }
|
||||
|
||||
func (*assembledSpellExtractor) ReferenceSlots() []contracts.ReferenceSlot { return nil }
|
||||
|
||||
@@ -8,6 +8,8 @@ import (
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/buildinfo"
|
||||
)
|
||||
|
||||
func TestCommandHelpSpellingsWriteUsageToStdout(t *testing.T) {
|
||||
@@ -38,6 +40,7 @@ func TestCommandSyntaxErrorsUseStderrAndExitTwo(t *testing.T) {
|
||||
{name: "unknown pipelines subcommand", args: []string{"pipelines", "unknown"}, want: "unknown pipelines subcommand"},
|
||||
{name: "malformed run flag", args: []string{"run", "demo", "--chunk_cache", "invalid"}, want: "not supported"},
|
||||
{name: "unknown flag", args: []string{"config", "validate", "--unknown"}, want: "flag provided but not defined"},
|
||||
{name: "version arguments", args: []string{"--version", "extra"}, want: "--version does not accept arguments"},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
@@ -50,6 +53,33 @@ func TestCommandSyntaxErrorsUseStderrAndExitTwo(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestCommandVersionOutput(t *testing.T) {
|
||||
previous := buildinfo.Override
|
||||
t.Cleanup(func() { buildinfo.Override = previous })
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
override string
|
||||
wantCode int
|
||||
wantStdout string
|
||||
wantStderr string
|
||||
}{
|
||||
{name: "development", wantStdout: "notarius development\n"},
|
||||
{name: "release override", override: "v1.2.3", wantStdout: "notarius v1.2.3\n"},
|
||||
{name: "invalid override", override: "release", wantCode: 1, wantStderr: "notarius: build version override is not a stable release tag\n"},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
buildinfo.Override = tt.override
|
||||
var stdout, stderr bytes.Buffer
|
||||
code := RunWithOptions([]string{"--version"}, &stdout, &stderr, Options{})
|
||||
if code != tt.wantCode || stdout.String() != tt.wantStdout || stderr.String() != tt.wantStderr {
|
||||
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestConfigDiscoveryPrefersExplicitPathThenEnvironment(t *testing.T) {
|
||||
explicit := writeCommandConfig(t, "explicit", "alpha")
|
||||
environment := writeCommandConfig(t, "environment", "beta")
|
||||
|
||||
@@ -189,10 +189,18 @@ func TestMaintainedCompleteExamplePublishesRegistryBackedEntityOccurrences(t *te
|
||||
}
|
||||
|
||||
evidence := readProductionJSON[evidencecontext.Document](t, filepath.Join(runRoot, "evidence-context.json"))
|
||||
for _, laneID := range []string{"enemy-events", "npc-registry", "npc-occurrences", "item-registry", "item-occurrences", "location-registry", "location-occurrences"} {
|
||||
if !containsString(evidence.SelectedLanes, laneID) || !evidenceHasLane(evidence, laneID) {
|
||||
t.Fatalf("evidence context = %#v, want direct %s evidence", evidence, laneID)
|
||||
if len(evidence) == 0 {
|
||||
t.Fatalf("evidence context = %#v, want selected source-unit evidence", evidence)
|
||||
}
|
||||
seenEvidenceUnits := make(map[int]struct{}, len(evidence))
|
||||
for _, unit := range evidence {
|
||||
if unit.Ref.SourceID != "session-ravenfall" || unit.Ref.StartUnitID != unit.ID || unit.Ref.EndUnitID != unit.ID {
|
||||
t.Fatalf("evidence unit = %#v, want unchanged source-unit self-reference", unit)
|
||||
}
|
||||
if _, exists := seenEvidenceUnits[unit.ID]; exists {
|
||||
t.Fatalf("evidence context = %#v, want each source unit once", evidence)
|
||||
}
|
||||
seenEvidenceUnits[unit.ID] = struct{}{}
|
||||
}
|
||||
|
||||
requests := client.requestsFor(enemyevents.PromptID)
|
||||
@@ -219,14 +227,15 @@ func TestMaintainedCompleteExamplePublishesRegistryBackedEntityOccurrences(t *te
|
||||
}
|
||||
for _, request := range locationRequests {
|
||||
registryInput := request.Inputs["location_registry"]
|
||||
if !strings.Contains(string(registryInput.Content), "Moon Gate") || !strings.Contains(string(registryInput.Content), `"id"`) || strings.Contains(string(registryInput.Content), "source_refs") {
|
||||
t.Fatalf("location occurrence registry input = %q, want source-free ID grounding", registryInput.Content)
|
||||
if !strings.Contains(string(registryInput.Content), "Moon Gate") || !strings.Contains(string(registryInput.Content), "registry_refs") || strings.Contains(string(registryInput.Content), `"id"`) || strings.Contains(string(registryInput.Content), "source_refs") {
|
||||
t.Fatalf("location occurrence registry input = %q, want contextual selector grounding", registryInput.Content)
|
||||
}
|
||||
}
|
||||
for _, test := range []struct {
|
||||
promptID string
|
||||
slot string
|
||||
name string
|
||||
promptID string
|
||||
slot string
|
||||
name string
|
||||
requiresIDs bool
|
||||
}{
|
||||
{promptID: npcoccurrences.PromptID, slot: "npc_registry", name: "Kesh"},
|
||||
{promptID: itemoccurrences.PromptID, slot: "item_registry", name: "Moonblade"},
|
||||
@@ -237,8 +246,9 @@ func TestMaintainedCompleteExamplePublishesRegistryBackedEntityOccurrences(t *te
|
||||
}
|
||||
for _, request := range requests {
|
||||
registryInput := request.Inputs[test.slot]
|
||||
if !strings.Contains(string(registryInput.Content), test.name) || !strings.Contains(string(registryInput.Content), `"id"`) || strings.Contains(string(registryInput.Content), "source_refs") {
|
||||
t.Fatalf("%s registry input = %q, want source-free ID grounding", test.promptID, registryInput.Content)
|
||||
hasID := strings.Contains(string(registryInput.Content), `"id"`)
|
||||
if !strings.Contains(string(registryInput.Content), test.name) || hasID != test.requiresIDs || strings.Contains(string(registryInput.Content), "source_refs") {
|
||||
t.Fatalf("%s registry input = %q, want source-free configured grounding", test.promptID, registryInput.Content)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -319,16 +329,16 @@ func (client *enemyEventLLMClient) CompleteStructured(ctx context.Context, reque
|
||||
} else {
|
||||
var registry struct {
|
||||
Items []struct {
|
||||
ID string `json:"id"`
|
||||
Name string `json:"name"`
|
||||
} `json:"items"`
|
||||
}
|
||||
if err := json.Unmarshal(request.Inputs["item_registry"].Content, ®istry); err != nil {
|
||||
return contracts.StructuredCompletionResponse{}, fmt.Errorf("decode generated item registry: %w", err)
|
||||
}
|
||||
if len(registry.Items) != 1 {
|
||||
if len(registry.Items) != 1 || registry.Items[0].Name != "Moonblade" {
|
||||
return contracts.StructuredCompletionResponse{}, fmt.Errorf("generated item registry has %d items, want 1", len(registry.Items))
|
||||
}
|
||||
content = []byte(fmt.Sprintf(`{"occurrences":[{"item_id":%q,"name":"Moonblade","kind":"discovered","quantity":null,"from":null,"to":null,"source_refs":[{"start_segment":5,"end_segment":5}]}]}`, registry.Items[0].ID))
|
||||
content = []byte(`{"occurrences":[{"name":"Moonblade","kind":"discovered","quantity":null,"from":null,"to":null,"source_refs":[{"start_unit_id":5,"end_unit_id":5}]}]}`)
|
||||
}
|
||||
case combat.PromptID:
|
||||
content = []byte(`{"combat_turns":[{"actor":"Kesh","turn_kind":"turn","source_refs":[{"start_unit_id":8,"end_unit_id":8}]}]}`)
|
||||
@@ -336,23 +346,27 @@ func (client *enemyEventLLMClient) CompleteStructured(ctx context.Context, reque
|
||||
if combatScene {
|
||||
var registry struct {
|
||||
NPCs []struct {
|
||||
ID string `json:"id"`
|
||||
Name string `json:"name"`
|
||||
} `json:"npcs"`
|
||||
}
|
||||
if err := json.Unmarshal(request.Inputs["npc_registry"].Content, ®istry); err != nil {
|
||||
return contracts.StructuredCompletionResponse{}, fmt.Errorf("decode generated NPC registry: %w", err)
|
||||
}
|
||||
if len(registry.NPCs) == 0 {
|
||||
if len(registry.NPCs) == 0 || registry.NPCs[0].Name != "Kesh" {
|
||||
return contracts.StructuredCompletionResponse{}, fmt.Errorf("generated NPC registry has no NPCs")
|
||||
}
|
||||
content = []byte(fmt.Sprintf(`{"occurrences":[{"npc_id":%q,"name":"Kesh","kind":"combat_opponent","source_refs":[{"start_unit_id":7,"end_unit_id":7}]}]}`, registry.NPCs[0].ID))
|
||||
content = []byte(`{"occurrences":[{"name":"Kesh","kind":"combat_opponent","source_refs":[{"start_unit_id":7,"end_unit_id":7}]}]}`)
|
||||
} else {
|
||||
content = []byte(`{"occurrences":[]}`)
|
||||
}
|
||||
case locationoccurrences.PromptID:
|
||||
var registry struct {
|
||||
Locations []struct {
|
||||
ID string `json:"id"`
|
||||
Name string `json:"name"`
|
||||
RegistryRefs []struct {
|
||||
StartUnitID int `json:"start_unit_id"`
|
||||
EndUnitID int `json:"end_unit_id"`
|
||||
} `json:"registry_refs"`
|
||||
} `json:"locations"`
|
||||
}
|
||||
if err := json.Unmarshal(request.Inputs["location_registry"].Content, ®istry); err != nil {
|
||||
@@ -362,14 +376,18 @@ func (client *enemyEventLLMClient) CompleteStructured(ctx context.Context, reque
|
||||
return contracts.StructuredCompletionResponse{}, fmt.Errorf("generated location registry has no locations")
|
||||
}
|
||||
unitID := 1
|
||||
locationID := registry.Locations[0].ID
|
||||
location := registry.Locations[0]
|
||||
if combatScene {
|
||||
unitID = 7
|
||||
if len(registry.Locations) > 1 {
|
||||
locationID = registry.Locations[1].ID
|
||||
location = registry.Locations[1]
|
||||
}
|
||||
}
|
||||
content = []byte(fmt.Sprintf(`{"occurrences":[{"location_id":%q,"name":"Moon Gate","kind":"visited","source_refs":[{"start_unit_id":%d,"end_unit_id":%d}]}]}`, locationID, unitID, unitID))
|
||||
registryRefs, err := json.Marshal(location.RegistryRefs)
|
||||
if err != nil {
|
||||
return contracts.StructuredCompletionResponse{}, fmt.Errorf("encode location selector: %w", err)
|
||||
}
|
||||
content = []byte(fmt.Sprintf(`{"occurrences":[{"name":%q,"registry_refs":%s,"kind":"visited","source_refs":[{"start_unit_id":%d,"end_unit_id":%d}]}]}`, location.Name, registryRefs, unitID, unitID))
|
||||
case enemyevents.PromptID:
|
||||
content = []byte(`{"events":[{"name":"Kesh","kind":"fled","source_refs":[{"start_unit_id":10,"end_unit_id":10}]}]}`)
|
||||
default:
|
||||
@@ -378,8 +396,12 @@ func (client *enemyEventLLMClient) CompleteStructured(ctx context.Context, reque
|
||||
if err := json.Unmarshal(content, output); err != nil {
|
||||
return contracts.StructuredCompletionResponse{}, fmt.Errorf("populate fake structured target: %w", err)
|
||||
}
|
||||
snapshot, err := contracts.CloneStructuredCompletionRequest(request)
|
||||
if err != nil {
|
||||
return contracts.StructuredCompletionResponse{}, fmt.Errorf("clone fake request: %w", err)
|
||||
}
|
||||
client.mu.Lock()
|
||||
client.requests = append(client.requests, request)
|
||||
client.requests = append(client.requests, snapshot)
|
||||
client.mu.Unlock()
|
||||
return contracts.StructuredCompletionResponse{Content: content, Provider: "test", Model: "deterministic", ProfileID: request.ProfileID}, nil
|
||||
}
|
||||
@@ -390,7 +412,11 @@ func (client *enemyEventLLMClient) requestsFor(promptID string) []contracts.Stru
|
||||
var requests []contracts.StructuredCompletionRequest
|
||||
for _, request := range client.requests {
|
||||
if request.PromptID == promptID {
|
||||
requests = append(requests, request)
|
||||
snapshot, err := contracts.CloneStructuredCompletionRequest(request)
|
||||
if err != nil {
|
||||
panic(err)
|
||||
}
|
||||
requests = append(requests, snapshot)
|
||||
}
|
||||
}
|
||||
return requests
|
||||
@@ -405,17 +431,6 @@ func containsString(values []string, want string) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
func evidenceHasLane(value evidencecontext.Document, laneID string) bool {
|
||||
for _, context := range value.Contexts {
|
||||
for _, reference := range context.EvidenceRefs {
|
||||
if reference.LaneID == laneID {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func generatedReferenceBinding(bindings []pipeline.ReferenceBinding, slotName string) (pipeline.ReferenceBinding, bool) {
|
||||
for _, binding := range bindings {
|
||||
if binding.SlotName == slotName && binding.Artifact != nil {
|
||||
|
||||
@@ -19,12 +19,15 @@ import (
|
||||
"testing/fstest"
|
||||
"time"
|
||||
|
||||
"gopkg.in/yaml.v3"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/chunkmap"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/chunk/scenes"
|
||||
combatcodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/combatturns"
|
||||
@@ -311,6 +314,81 @@ func TestProductionPromptAssetsPrepareWithoutProviderCredentials(t *testing.T) {
|
||||
if _, err := pipeline.Prepare(effective.ResolvedPipeline, components.registries, pipeline.ModuleDependencies{LLM: &productionFakeLLMClient{}}); err != nil {
|
||||
t.Fatalf("prepare production scene and spell modules: %v", err)
|
||||
}
|
||||
|
||||
schemaFS, err := components.assets.SchemaFS()
|
||||
if err != nil {
|
||||
t.Fatalf("production schema assets: %v", err)
|
||||
}
|
||||
if _, err := fs.ReadFile(schemaFS, filepath.Base(semanticreconcile.SchemaAssetPath)); err != nil {
|
||||
t.Fatalf("generic reconciliation schema asset: %v", err)
|
||||
}
|
||||
options, err := components.assets.PromptKitOptions()
|
||||
if err != nil {
|
||||
t.Fatalf("production PromptKit options: %v", err)
|
||||
}
|
||||
options = append(options, promptkit.WithProfiles(promptkit.OpenAICompatibleProfile(promptkit.OpenAICompatibleProfileConfig{
|
||||
ID: "assembled-prompt-test", Endpoint: "http://127.0.0.1:1/v1", Model: "test",
|
||||
})))
|
||||
engine, err := promptkit.NewEngine(promptkit.Config{}, options...)
|
||||
if err != nil {
|
||||
t.Fatalf("production prompt engine: %v", err)
|
||||
}
|
||||
promptFS, err := components.assets.PromptFS()
|
||||
if err != nil {
|
||||
t.Fatalf("production prompt assets: %v", err)
|
||||
}
|
||||
type manifest struct {
|
||||
ID string `yaml:"id"`
|
||||
Version string `yaml:"version"`
|
||||
Inputs []struct {
|
||||
Name string `yaml:"name"`
|
||||
} `yaml:"inputs"`
|
||||
Messages []struct {
|
||||
Role string `yaml:"role"`
|
||||
} `yaml:"messages"`
|
||||
}
|
||||
preparedPrompts := 0
|
||||
if err := fs.WalkDir(promptFS, ".", func(path string, entry fs.DirEntry, walkErr error) error {
|
||||
if walkErr != nil {
|
||||
return walkErr
|
||||
}
|
||||
if entry.IsDir() || filepath.Base(path) != "prompt.yaml" {
|
||||
return nil
|
||||
}
|
||||
data, err := fs.ReadFile(promptFS, path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
var prompt manifest
|
||||
if err := yaml.Unmarshal(data, &prompt); err != nil {
|
||||
return err
|
||||
}
|
||||
for _, message := range prompt.Messages {
|
||||
if message.Role != promptkit.RoleSystem && message.Role != promptkit.RoleUser {
|
||||
return fmt.Errorf("production prompt %q uses role %q, want system or user", prompt.ID, message.Role)
|
||||
}
|
||||
}
|
||||
inputs := make(map[string]promptkit.ArtifactRef, len(prompt.Inputs))
|
||||
for _, input := range prompt.Inputs {
|
||||
inputs[input.Name] = promptkit.Inline(`{}`)
|
||||
}
|
||||
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{
|
||||
PromptID: prompt.ID, PromptVersion: prompt.Version, ProfileID: "assembled-prompt-test", Inputs: inputs,
|
||||
})
|
||||
if err != nil {
|
||||
return fmt.Errorf("prepare production prompt %q: %w", prompt.ID, err)
|
||||
}
|
||||
if prepared.OutputContract.RepairAttempts != 1 {
|
||||
return fmt.Errorf("prompt %q repair attempts = %d, want 1", prompt.ID, prepared.OutputContract.RepairAttempts)
|
||||
}
|
||||
preparedPrompts++
|
||||
return nil
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if preparedPrompts == 0 {
|
||||
t.Fatal("prepared no production prompts")
|
||||
}
|
||||
}
|
||||
|
||||
func TestProductionSpellValidatorsPrepareFromMaterializedCatalog(t *testing.T) {
|
||||
@@ -1021,8 +1099,12 @@ func (client *productionFakeLLMClient) CompleteStructured(ctx context.Context, r
|
||||
if err := json.Unmarshal(content, out); err != nil {
|
||||
return contracts.StructuredCompletionResponse{}, fmt.Errorf("populate fake structured target: %w", err)
|
||||
}
|
||||
snapshot, err := contracts.CloneStructuredCompletionRequest(req)
|
||||
if err != nil {
|
||||
return contracts.StructuredCompletionResponse{}, fmt.Errorf("clone fake request: %w", err)
|
||||
}
|
||||
client.mu.Lock()
|
||||
client.requests = append(client.requests, req)
|
||||
client.requests = append(client.requests, snapshot)
|
||||
client.mu.Unlock()
|
||||
return contracts.StructuredCompletionResponse{Content: content, Provider: "test", Model: "deterministic", ProfileID: req.ProfileID}, nil
|
||||
}
|
||||
@@ -1033,7 +1115,11 @@ func (client *productionFakeLLMClient) requestsFor(promptID string) []contracts.
|
||||
var requests []contracts.StructuredCompletionRequest
|
||||
for _, req := range client.requests {
|
||||
if req.PromptID == promptID {
|
||||
requests = append(requests, req)
|
||||
snapshot, err := contracts.CloneStructuredCompletionRequest(req)
|
||||
if err != nil {
|
||||
panic(err)
|
||||
}
|
||||
requests = append(requests, snapshot)
|
||||
}
|
||||
}
|
||||
return requests
|
||||
|
||||
@@ -71,10 +71,10 @@ api_key_env: NOTARIUS_PROMPTKIT_PROFILE_INSPECTION_TEST_KEY
|
||||
},
|
||||
{
|
||||
name: "malformed profile",
|
||||
profilePath: writeProfile(t, "malformed-profile", "id: malformed-profile\nbackend: [\n"),
|
||||
profilePath: writeProfile(t, "malformed-profile", "id: malformed-profile\nendpoint: https://provider.example/v1?credential=forbidden\nmodel: malformed-model\n"),
|
||||
profileID: "malformed-profile",
|
||||
wantErr: []string{`PromptKit profile "malformed-profile" is invalid or unreadable`},
|
||||
rejectErr: []string{"malformed-profile.yaml", "backend: ["},
|
||||
rejectErr: []string{"malformed-profile.yaml", "credential=forbidden"},
|
||||
},
|
||||
{
|
||||
name: "invalid profile source",
|
||||
@@ -157,3 +157,25 @@ func TestExplicitPromptKitProfileValidationUsesFallbackAssets(t *testing.T) {
|
||||
t.Fatalf("validateExplicitPromptKitProfiles() error = %v, want nil", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestExplicitPromptKitProfileValidationRejectsInvalidInheritanceBeforeGeneration(t *testing.T) {
|
||||
for _, profiles := range []string{
|
||||
"id: child\nbase_profile: missing\n",
|
||||
"id: first\nbase_profile: second\n\n---\nid: second\nbase_profile: first\n",
|
||||
} {
|
||||
t.Run("invalid inheritance", func(t *testing.T) {
|
||||
profilePath := filepath.Join(t.TempDir(), "profiles.yaml")
|
||||
if err := os.WriteFile(profilePath, []byte(profiles), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
profileID := "child"
|
||||
if strings.Contains(profiles, "id: first") {
|
||||
profileID = "first"
|
||||
}
|
||||
err := validateExplicitPromptKitProfiles(context.Background(), config.Config{PromptKit: config.PromptKitConfig{ProfileFile: profilePath}}, []string{profileID}, nil)
|
||||
if err == nil || !strings.Contains(err.Error(), "invalid or unreadable") || strings.Contains(err.Error(), profilePath) {
|
||||
t.Fatalf("profile preflight error = %v", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -400,6 +400,9 @@ func (referenceContractCodecA) Encode(stateTestArtifact) ([]byte, error) {
|
||||
func (referenceContractCodecA) Decode([]byte) (stateTestArtifact, error) {
|
||||
return stateTestArtifact{Value: "ok"}, nil
|
||||
}
|
||||
func (codec referenceContractCodecA) DecodeCandidate(content []byte) (stateTestArtifact, error) {
|
||||
return codec.Decode(content)
|
||||
}
|
||||
|
||||
func (referenceContractCodecB) Kind() contracts.ArtifactKind { return referenceContractKindBeta }
|
||||
func (referenceContractCodecB) Schema() contracts.ArtifactSchema {
|
||||
@@ -415,6 +418,9 @@ func (referenceContractCodecB) Encode(stateTestArtifact) ([]byte, error) {
|
||||
func (referenceContractCodecB) Decode([]byte) (stateTestArtifact, error) {
|
||||
return stateTestArtifact{Value: "ok"}, nil
|
||||
}
|
||||
func (codec referenceContractCodecB) DecodeCandidate(content []byte) (stateTestArtifact, error) {
|
||||
return codec.Decode(content)
|
||||
}
|
||||
|
||||
func referenceContractLane(t *testing.T, resolved pipeline.ResolvedPipeline, id string) pipeline.ResolvedArtifactLane {
|
||||
t.Helper()
|
||||
|
||||
@@ -16,9 +16,11 @@ import (
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/buildinfo"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/debugbundle"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/fileio"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/checkpoint"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/chunkplan"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
@@ -30,6 +32,7 @@ import (
|
||||
const defaultConfigPath = "/usr/local/etc/notarius/config.yml"
|
||||
const usage = `Usage:
|
||||
notarius help
|
||||
notarius --version
|
||||
notarius run <pipeline-id> --input path/to/source.json [--json] [flags]
|
||||
notarius config validate --config path/to/config.yml [--pipeline pipeline-id] [--only lane-a,lane-b]
|
||||
notarius pipelines list --config path/to/config.yml [--json]
|
||||
@@ -61,6 +64,20 @@ func Run(args []string, stdout, stderr io.Writer) int {
|
||||
}
|
||||
|
||||
func RunWithOptions(args []string, stdout, stderr io.Writer, opts Options) int {
|
||||
if len(args) > 0 && args[0] == "--version" {
|
||||
if len(args) != 1 {
|
||||
fmt.Fprintln(stderr, "notarius: --version does not accept arguments")
|
||||
return 2
|
||||
}
|
||||
version, err := buildinfo.Version()
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "notarius: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
fmt.Fprintf(stdout, "notarius %s\n", version)
|
||||
return 0
|
||||
}
|
||||
|
||||
var err error
|
||||
opts, err = normalizeOptions(opts)
|
||||
if err != nil {
|
||||
@@ -147,7 +164,7 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
|
||||
machineOutput := fs.Bool("json", false, "write the successful run result as JSON")
|
||||
debug := fs.Bool("debug", false, "write a debug bundle")
|
||||
debugDir := fs.String("debug-dir", "", "debug bundle directory")
|
||||
llmProfile := fs.String("llm-profile", "", "LLM profile override")
|
||||
llmProfile := singleValueFlag{name: "--llm-profile"}
|
||||
reasoningEffort := singleValueFlag{name: "--reasoning-effort"}
|
||||
clearReasoningEffort := fs.Bool("clear-reasoning-effort", false, "clear the LLM profile reasoning effort")
|
||||
resume := fs.Bool("resume", false, "reuse compatible recorded checkpoints")
|
||||
@@ -157,6 +174,7 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
|
||||
referenceFlags := stringListFlag{}
|
||||
withoutReferenceFlags := stringListFlag{}
|
||||
fs.Var(&requestedSessionID, "session-id", "prompt session identifier")
|
||||
fs.Var(&llmProfile, "llm-profile", "LLM profile override")
|
||||
fs.Var(&reasoningEffort, "reasoning-effort", "reasoning effort override")
|
||||
fs.Var(&chunkCache, "chunk_cache", "chunk plan cache mode: auto, bypass, or refresh")
|
||||
fs.Var(&referenceFlags, "reference", "reference binding, as slot=path, chunk.slot=path, merge.slot=path, lane.slot=path, lane.extract.slot=path, lane.merge.slot=path, or lane.normalize.slot=path")
|
||||
@@ -203,6 +221,10 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
|
||||
fmt.Fprintln(stderr, "notarius: --session-id must not be empty")
|
||||
return 2
|
||||
}
|
||||
if llmProfile.set && strings.TrimSpace(llmProfile.value) == "" {
|
||||
fmt.Fprintln(stderr, "notarius: --llm-profile must not be empty")
|
||||
return 2
|
||||
}
|
||||
if reasoningEffort.set && *clearReasoningEffort {
|
||||
fmt.Fprintln(stderr, "notarius: --reasoning-effort cannot be combined with --clear-reasoning-effort")
|
||||
return 2
|
||||
@@ -338,7 +360,7 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
|
||||
PipelineID: pipelineID,
|
||||
Only: only,
|
||||
Catalog: catalog,
|
||||
LLMProfileOverride: *llmProfile,
|
||||
LLMProfileOverride: strings.TrimSpace(llmProfile.value),
|
||||
ReferenceOverrides: referenceOverrides,
|
||||
ReferenceUnbinds: referenceUnbinds,
|
||||
})
|
||||
@@ -426,7 +448,7 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
|
||||
if err != nil {
|
||||
return failPipelineCommand(stderr, commandState, terminalWriter, err)
|
||||
}
|
||||
checkpointRecorder, checkpointLoader, err := checkpointHandlersForRun(effective.Config.Cache.Checkpoints, opts, effective.ResolvedPipeline, prepared.CheckpointFingerprints(), llmFingerprints, rawInput, only, llmProfiles, strings.TrimSpace(*llmProfile), effectiveSessionID, runtimeOverrides, *resume)
|
||||
checkpointRecorder, checkpointLoader, err := checkpointHandlersForRun(effective.Config.Cache.Checkpoints, opts, effective.ResolvedPipeline, prepared.CheckpointFingerprints(), llmFingerprints, rawInput, only, llmProfiles, strings.TrimSpace(llmProfile.value), effectiveSessionID, runtimeOverrides, *resume)
|
||||
if err != nil {
|
||||
return failPipelineCommand(stderr, commandState, terminalWriter, err)
|
||||
}
|
||||
@@ -752,17 +774,10 @@ func configSource(configPath string) string {
|
||||
}
|
||||
|
||||
func writeOutputFiles(runOutputDir string, files []contracts.OutputFile) error {
|
||||
type outputTarget struct {
|
||||
path string
|
||||
file contracts.OutputFile
|
||||
}
|
||||
targets := make([]outputTarget, 0, len(files))
|
||||
for _, file := range files {
|
||||
targetPath, err := outputFilePath(runOutputDir, file.Name)
|
||||
if err != nil {
|
||||
if _, err := outputFilePath(runOutputDir, file.Name); err != nil {
|
||||
return err
|
||||
}
|
||||
targets = append(targets, outputTarget{path: targetPath, file: file})
|
||||
}
|
||||
|
||||
outputParent := filepath.Dir(runOutputDir)
|
||||
@@ -775,12 +790,9 @@ func writeOutputFiles(runOutputDir string, files []contracts.OutputFile) error {
|
||||
}
|
||||
return fmt.Errorf("create output run directory %q: %w", runOutputDir, err)
|
||||
}
|
||||
for _, target := range targets {
|
||||
if err := os.MkdirAll(filepath.Dir(target.path), 0o755); err != nil {
|
||||
return fmt.Errorf("create output directory %q: %w", filepath.Dir(target.path), err)
|
||||
}
|
||||
if err := writeFileAtomic(target.path, target.file.Bytes, 0o644); err != nil {
|
||||
return fmt.Errorf("write output file %q: %w", target.file.Name, err)
|
||||
for _, file := range files {
|
||||
if err := fileio.WriteBytes(runOutputDir, file.Name, file.Bytes, 0o755, 0o644); err != nil {
|
||||
return fmt.Errorf("write output file %q: %w", file.Name, err)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
@@ -823,38 +835,6 @@ func outputFilePath(runOutputDir, logicalName string) (string, error) {
|
||||
return target, nil
|
||||
}
|
||||
|
||||
func writeFileAtomic(path string, data []byte, perm os.FileMode) error {
|
||||
dir := filepath.Dir(path)
|
||||
temp, err := os.CreateTemp(dir, "."+filepath.Base(path)+".tmp-*")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
tempPath := temp.Name()
|
||||
removeTemp := true
|
||||
defer func() {
|
||||
if removeTemp {
|
||||
_ = os.Remove(tempPath)
|
||||
}
|
||||
}()
|
||||
|
||||
if _, err := temp.Write(data); err != nil {
|
||||
_ = temp.Close()
|
||||
return err
|
||||
}
|
||||
if err := temp.Chmod(perm); err != nil {
|
||||
_ = temp.Close()
|
||||
return err
|
||||
}
|
||||
if err := temp.Close(); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.Rename(tempPath, path); err != nil {
|
||||
return err
|
||||
}
|
||||
removeTemp = false
|
||||
return nil
|
||||
}
|
||||
|
||||
func reorderRunArgs(args []string) []string {
|
||||
var flags []string
|
||||
var positionals []string
|
||||
|
||||
@@ -43,6 +43,12 @@ func TestRunControlsRejectSyntaxWithoutAllocatingState(t *testing.T) {
|
||||
{name: "blank session ID", args: func(roots stateTestRoots) []string {
|
||||
return []string{"run", "sample", "--config", roots.config, "--input", roots.input, "--session-id", ""}
|
||||
}},
|
||||
{name: "blank LLM profile", args: func(roots stateTestRoots) []string {
|
||||
return []string{"run", "sample", "--config", roots.config, "--input", roots.input, "--llm-profile", ""}
|
||||
}},
|
||||
{name: "whitespace LLM profile", args: func(roots stateTestRoots) []string {
|
||||
return []string{"run", "sample", "--config", roots.config, "--input", roots.input, "--llm-profile", " \t "}
|
||||
}},
|
||||
{name: "multiple pipeline IDs", args: func(roots stateTestRoots) []string {
|
||||
return []string{"run", "sample", "extra", "--config", roots.config, "--input", roots.input}
|
||||
}},
|
||||
@@ -253,7 +259,7 @@ func TestRunLLMProfileOverrideAndValidationUseInjectedBoundaries(t *testing.T) {
|
||||
return nil, nil, nil
|
||||
}
|
||||
var stdout, stderr bytes.Buffer
|
||||
code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass", "--llm-profile", "override-profile"}, &stdout, &stderr, opts)
|
||||
code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass", "--llm-profile", " override-profile "}, &stdout, &stderr, opts)
|
||||
if code != 0 || stderr.Len() != 0 {
|
||||
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||
}
|
||||
|
||||
@@ -38,10 +38,36 @@ func TestWriteOutputFilesSupportsNestedLogicalPaths(t *testing.T) {
|
||||
if err := writeOutputFiles(runPath, []contracts.OutputFile{{Name: "nested/result.json", Bytes: []byte("result")}}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
data, err := os.ReadFile(filepath.Join(runPath, "nested", "result.json"))
|
||||
resultPath := filepath.Join(runPath, "nested", "result.json")
|
||||
data, err := os.ReadFile(resultPath)
|
||||
if err != nil || string(data) != "result" {
|
||||
t.Fatalf("nested output = %q, %v", data, err)
|
||||
}
|
||||
for path, want := range map[string]os.FileMode{runPath: 0o755, filepath.Join(runPath, "nested"): 0o755, resultPath: 0o644} {
|
||||
info, err := os.Stat(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
mode := info.Mode().Perm()
|
||||
if mode&^want != 0 {
|
||||
t.Fatalf("%s mode = %#o, must not be broader than %#o", path, mode, want)
|
||||
}
|
||||
if info.IsDir() && mode&0o700 != 0o700 {
|
||||
t.Fatalf("%s mode = %#o, want owner access", path, mode)
|
||||
}
|
||||
if !info.IsDir() && mode != want {
|
||||
t.Fatalf("%s mode = %#o, want %#o", path, mode, want)
|
||||
}
|
||||
}
|
||||
entries, err := os.ReadDir(filepath.Join(runPath, "nested"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, entry := range entries {
|
||||
if strings.Contains(entry.Name(), ".tmp-") {
|
||||
t.Fatalf("temporary file remains: %s", entry.Name())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriteOutputFilesRejectsUnsafeNamesBeforeAllocatingRunDirectory(t *testing.T) {
|
||||
@@ -75,7 +101,7 @@ func TestWriteOutputFilesRetainsNewPartialDirectoryAndPreservesSibling(t *testin
|
||||
{Name: "blocked", Bytes: []byte("partial output")},
|
||||
{Name: "blocked/nested.json", Bytes: []byte("unreachable")},
|
||||
})
|
||||
if err == nil || !strings.Contains(err.Error(), "create output directory") {
|
||||
if err == nil || !strings.Contains(err.Error(), `write output file "blocked/nested.json"`) {
|
||||
t.Fatalf("writeOutputFiles() error = %v, want later directory failure", err)
|
||||
}
|
||||
if got, err := os.ReadFile(filepath.Join(runPath, "blocked")); err != nil || string(got) != "partial output" {
|
||||
|
||||
@@ -7,22 +7,24 @@ import (
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
)
|
||||
|
||||
const runResultSchemaVersion = "notarius.run-result.v1"
|
||||
|
||||
type runResult struct {
|
||||
SchemaVersion string `json:"schema_version"`
|
||||
RunID string `json:"run_id"`
|
||||
PipelineID string `json:"pipeline_id"`
|
||||
OutputDirectory string `json:"output_directory"`
|
||||
IndexFile string `json:"index_file,omitempty"`
|
||||
NormalizedOutputCount int `json:"normalized_output_count"`
|
||||
RejectedOutputCount int `json:"rejected_output_count"`
|
||||
WarningCount int `json:"warning_count"`
|
||||
ValidationStatus string `json:"validation_status"`
|
||||
DebugDirectory string `json:"debug_directory,omitempty"`
|
||||
SchemaVersion string `json:"schema_version"`
|
||||
RunID string `json:"run_id"`
|
||||
PipelineID string `json:"pipeline_id"`
|
||||
OutputDirectory string `json:"output_directory"`
|
||||
IndexFile string `json:"index_file,omitempty"`
|
||||
NormalizedOutputCount int `json:"normalized_output_count"`
|
||||
RejectedOutputCount int `json:"rejected_output_count"`
|
||||
WarningCount int `json:"warning_count"`
|
||||
ValidationStatus string `json:"validation_status"`
|
||||
ValidationSummaries []artifacts.ValidationSummary `json:"validation_summaries,omitempty"`
|
||||
DebugDirectory string `json:"debug_directory,omitempty"`
|
||||
}
|
||||
|
||||
func newRunResult(resolved pipeline.ResolvedPipeline, output pipeline.RunOutput, outputDirectory, debugDirectory string) (runResult, error) {
|
||||
@@ -59,6 +61,7 @@ func newRunResult(resolved pipeline.ResolvedPipeline, output pipeline.RunOutput,
|
||||
RejectedOutputCount: len(output.Rejected),
|
||||
WarningCount: len(output.Warnings),
|
||||
ValidationStatus: output.Manifest.ValidationStatus,
|
||||
ValidationSummaries: cloneValidationSummaries(output.Manifest.ValidationSummaries),
|
||||
}
|
||||
|
||||
if strings.TrimSpace(debugDirectory) != "" {
|
||||
@@ -85,6 +88,17 @@ func newRunResult(resolved pipeline.ResolvedPipeline, output pipeline.RunOutput,
|
||||
return result, nil
|
||||
}
|
||||
|
||||
func cloneValidationSummaries(summaries []artifacts.ValidationSummary) []artifacts.ValidationSummary {
|
||||
if len(summaries) == 0 {
|
||||
return nil
|
||||
}
|
||||
cloned := make([]artifacts.ValidationSummary, len(summaries))
|
||||
for index, summary := range summaries {
|
||||
cloned[index] = artifacts.CloneValidationSummary(summary)
|
||||
}
|
||||
return cloned
|
||||
}
|
||||
|
||||
func encodeRunResult(result runResult) ([]byte, error) {
|
||||
encoded, err := json.Marshal(result)
|
||||
if err != nil {
|
||||
|
||||
@@ -92,7 +92,7 @@ func TestRunResultReportsSuccessfulRejection(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
configBytes = []byte(replaceRequiredOnce(t, string(configBytes), " normalize: test/normalize\n", " normalize:\n module: test/normalize\n validators:\n - generic/always_reject\n"))
|
||||
configBytes = []byte(replaceRequiredOnce(t, string(configBytes), " normalize: test/normalize\n", " normalize:\n module: test/normalize\n validators:\n - generic/always_reject\n validation_policy:\n semantic_rejection: reject_output\n"))
|
||||
if err := os.WriteFile(roots.config, configBytes, 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
@@ -55,6 +55,9 @@ func TestRunResultEncodesRequiredFieldsAndCounts(t *testing.T) {
|
||||
if got := decoded["warning_count"]; got != float64(1) {
|
||||
t.Fatalf("warning_count = %v", got)
|
||||
}
|
||||
if got := decoded["validation_summaries"]; got != nil {
|
||||
t.Fatalf("validation_summaries = %#v, want omitted when empty", got)
|
||||
}
|
||||
if got := decoded["output_directory"]; got != filepath.Join(mustWorkingDirectory(t), "relative-output") {
|
||||
t.Fatalf("output_directory = %q", got)
|
||||
}
|
||||
@@ -112,6 +115,26 @@ func TestRunResultOmitsIndexFileForOtherOutputModules(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunResultProjectsOwnedValidationSummaries(t *testing.T) {
|
||||
output := testRunOutput()
|
||||
output.Manifest.ValidationSummaries = []artifacts.ValidationSummary{{Status: "incomplete", IncompleteValidators: []string{"validator"}, ProducerAttemptCount: 1, TerminalAction: "warn_continue"}}
|
||||
result, err := newRunResult(testResolvedPipeline(pipeline.DefaultOutputModule), output, "output", "")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
output.Manifest.ValidationSummaries[0].IncompleteValidators[0] = "caller mutation"
|
||||
if got := result.ValidationSummaries[0].IncompleteValidators; len(got) != 1 || got[0] != "validator" {
|
||||
t.Fatalf("result validation summaries = %#v", result.ValidationSummaries)
|
||||
}
|
||||
encoded, err := encodeRunResult(result)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !bytes.Contains(encoded, []byte(`"validation_summaries":[{"status":"incomplete","incomplete_validators":["validator"],"producer_attempt_count":1,"terminal_action":"warn_continue"}]`)) {
|
||||
t.Fatalf("encoded result = %s", encoded)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunResultRequiresOneProductionIndexFile(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
|
||||
@@ -410,8 +410,8 @@ func TestMaintainedProductionOverlayRunAlignsGroundingValidationAndProvenance(t
|
||||
t.Fatalf("spell requests = %d, want one", len(requests))
|
||||
}
|
||||
catalogInput, ok := requests[0].Inputs[spellcatalog.SpellCatalogReferenceSlot]
|
||||
if !ok || !strings.Contains(string(catalogInput.Content), "Aegis of Emberfall") || strings.Contains(string(catalogInput.Content), "Emberfall Aegis") {
|
||||
t.Fatalf("spell catalog prompt input = %#v, want canonical overlay name without alias", catalogInput)
|
||||
if !ok || !strings.Contains(string(catalogInput.Content), `"canonical_name":"Aegis of Emberfall"`) || !strings.Contains(string(catalogInput.Content), `"aliases":["Emberfall Aegis"]`) {
|
||||
t.Fatalf("spell catalog prompt input = %#v, want canonical overlay name and recognition alias", catalogInput)
|
||||
}
|
||||
artifact := readProductionJSON[dnd.SpellList](t, filepath.Join(runRoot, "lanes", "spells.json"))
|
||||
if len(artifact.SpellCasts) != 1 || artifact.SpellCasts[0].Spell != "Aegis of Emberfall" {
|
||||
|
||||
@@ -65,6 +65,7 @@ func TestProductionSpellCatalogValidationRetries(t *testing.T) {
|
||||
t.Fatalf("materialize production references: %v", err)
|
||||
}
|
||||
materialized.Steps[0].ArtifactLanes[0].Extract.Retries = retries
|
||||
materialized.Steps[0].ArtifactLanes[0].ExtractValidationPolicy.SemanticRejection = pipeline.SemanticRejectionRejectOutput
|
||||
|
||||
llmClient := &catalogRetryLLMClient{responses: tt.responses}
|
||||
prepared, err := pipeline.Prepare(materialized, components.registries, pipeline.ModuleDependencies{LLM: llmClient})
|
||||
@@ -91,8 +92,8 @@ func TestProductionSpellCatalogValidationRetries(t *testing.T) {
|
||||
if rejection.ReasonCode != "unknown_spell" || rejection.AttemptCount != retries+1 {
|
||||
t.Fatalf("rejection = %#v, want exhausted unknown-spell rejection", rejection)
|
||||
}
|
||||
if len(output.Warnings) != 0 {
|
||||
t.Fatalf("warnings = %#v, want no warnings from rejected attempts", output.Warnings)
|
||||
if len(output.Warnings) != 1 || output.Warnings[0].ReasonCode != "spell_not_near_source" {
|
||||
t.Fatalf("warnings = %#v, want complete terminal validation warnings", output.Warnings)
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
@@ -955,6 +955,9 @@ func (stateTestCodec) Encode(v stateTestArtifact) ([]byte, error) {
|
||||
func (stateTestCodec) Decode([]byte) (stateTestArtifact, error) {
|
||||
return stateTestArtifact{Value: "ok"}, nil
|
||||
}
|
||||
func (codec stateTestCodec) DecodeCandidate(content []byte) (stateTestArtifact, error) {
|
||||
return codec.Decode(content)
|
||||
}
|
||||
|
||||
type stateTestExtractor struct{ harness *stateTestHarness }
|
||||
|
||||
|
||||
@@ -93,17 +93,44 @@ type NormalizedOutputManifest struct {
|
||||
}
|
||||
|
||||
type RejectedOutputManifest struct {
|
||||
Stage string `json:"stage"`
|
||||
StepID string `json:"step_id,omitempty"`
|
||||
LaneID string `json:"lane_id,omitempty"`
|
||||
ModuleKey string `json:"module_key,omitempty"`
|
||||
ChunkID string `json:"chunk_id,omitempty"`
|
||||
ChunkIndex int `json:"chunk_index,omitempty"`
|
||||
ValidatorName string `json:"validator_name,omitempty"`
|
||||
ReasonCode string `json:"reason_code,omitempty"`
|
||||
Message string `json:"message,omitempty"`
|
||||
AttemptCount int `json:"attempt_count,omitempty"`
|
||||
DiagnosticArtifactPath string `json:"diagnostic_artifact_path,omitempty"`
|
||||
Stage string `json:"stage"`
|
||||
StepID string `json:"step_id,omitempty"`
|
||||
LaneID string `json:"lane_id,omitempty"`
|
||||
ModuleKey string `json:"module_key,omitempty"`
|
||||
ChunkID string `json:"chunk_id,omitempty"`
|
||||
ChunkIndex int `json:"chunk_index,omitempty"`
|
||||
ValidatorName string `json:"validator_name,omitempty"`
|
||||
ReasonCode string `json:"reason_code,omitempty"`
|
||||
Message string `json:"message,omitempty"`
|
||||
AttemptCount int `json:"attempt_count,omitempty"`
|
||||
DiagnosticArtifactPath string `json:"diagnostic_artifact_path,omitempty"`
|
||||
Validation *ValidationSummary `json:"validation,omitempty"`
|
||||
}
|
||||
|
||||
// ValidationSummary is the bounded, durable outcome of validating one
|
||||
// producer result. It deliberately contains identities and stable codes, not
|
||||
// model responses, corrective guidance, validator diagnostics, or payloads.
|
||||
type ValidationSummary struct {
|
||||
Stage string `json:"stage,omitempty"`
|
||||
StepID string `json:"step_id,omitempty"`
|
||||
LaneID string `json:"lane_id,omitempty"`
|
||||
ModuleKey string `json:"module_key,omitempty"`
|
||||
ChunkID string `json:"chunk_id,omitempty"`
|
||||
ChunkIndex int `json:"chunk_index,omitempty"`
|
||||
Status string `json:"status"`
|
||||
RejectingValidators []string `json:"rejecting_validators,omitempty"`
|
||||
ReasonCodes []string `json:"reason_codes,omitempty"`
|
||||
IncompleteValidators []string `json:"incomplete_validators,omitempty"`
|
||||
ProducerAttemptCount int `json:"producer_attempt_count"`
|
||||
TerminalAction string `json:"terminal_action"`
|
||||
}
|
||||
|
||||
// CloneValidationSummary returns an independently owned durable summary.
|
||||
func CloneValidationSummary(summary ValidationSummary) ValidationSummary {
|
||||
summary.RejectingValidators = append([]string(nil), summary.RejectingValidators...)
|
||||
summary.ReasonCodes = append([]string(nil), summary.ReasonCodes...)
|
||||
summary.IncompleteValidators = append([]string(nil), summary.IncompleteValidators...)
|
||||
return summary
|
||||
}
|
||||
|
||||
type CheckpointDecisionManifest struct {
|
||||
@@ -163,6 +190,7 @@ type RunManifest struct {
|
||||
References []ReferenceProvenance `json:"references,omitempty"`
|
||||
NormalizedOutputs []NormalizedOutputManifest `json:"normalized_outputs,omitempty"`
|
||||
RejectedOutputs []RejectedOutputManifest `json:"rejected_outputs,omitempty"`
|
||||
ValidationSummaries []ValidationSummary `json:"validation_summaries,omitempty"`
|
||||
CheckpointDecisions []CheckpointDecisionManifest `json:"checkpoint_decisions,omitempty"`
|
||||
LLMProfiles []LLMProfileManifest `json:"llm_profiles,omitempty"`
|
||||
Metadata map[string]any `json:"metadata,omitempty"`
|
||||
|
||||
@@ -116,6 +116,11 @@ func (c *ConcurrencyConfig) recomputeStageWorkerDefaults() {
|
||||
|
||||
func clonePipelineProfile(in pipeline.PipelineProfile) pipeline.PipelineProfile {
|
||||
out := in
|
||||
out.ValidationPolicy = cloneValidationPolicyOverride(in.ValidationPolicy)
|
||||
if in.StructuredOutputRepairAttempts != nil {
|
||||
value := *in.StructuredOutputRepairAttempts
|
||||
out.StructuredOutputRepairAttempts = &value
|
||||
}
|
||||
out.Input = cloneModuleBinding(in.Input)
|
||||
out.Chunk = cloneModuleBinding(in.Chunk)
|
||||
out.Output = cloneModuleBinding(in.Output)
|
||||
@@ -196,6 +201,11 @@ func cloneReferenceSource(in pipeline.ReferenceSource) pipeline.ReferenceSource
|
||||
|
||||
func cloneModuleBinding(in pipeline.ModuleBinding) pipeline.ModuleBinding {
|
||||
out := in
|
||||
out.ValidationPolicy = cloneValidationPolicyOverride(in.ValidationPolicy)
|
||||
if in.StructuredOutputRepairAttempts != nil {
|
||||
value := *in.StructuredOutputRepairAttempts
|
||||
out.StructuredOutputRepairAttempts = &value
|
||||
}
|
||||
if len(in.Options) > 0 {
|
||||
out.Options = cloneOptions(in.Options)
|
||||
}
|
||||
@@ -204,6 +214,26 @@ func cloneModuleBinding(in pipeline.ModuleBinding) pipeline.ModuleBinding {
|
||||
return out
|
||||
}
|
||||
|
||||
func cloneValidationPolicyOverride(in *pipeline.ValidationPolicyOverride) *pipeline.ValidationPolicyOverride {
|
||||
if in == nil {
|
||||
return nil
|
||||
}
|
||||
out := *in
|
||||
if in.ProducerStructuralFailure != nil {
|
||||
value := *in.ProducerStructuralFailure
|
||||
out.ProducerStructuralFailure = &value
|
||||
}
|
||||
if in.SemanticRejection != nil {
|
||||
value := *in.SemanticRejection
|
||||
out.SemanticRejection = &value
|
||||
}
|
||||
if in.ValidatorFailure != nil {
|
||||
value := *in.ValidatorFailure
|
||||
out.ValidatorFailure = &value
|
||||
}
|
||||
return &out
|
||||
}
|
||||
|
||||
func cloneValidatorOverride(in pipeline.ValidatorOverride) pipeline.ValidatorOverride {
|
||||
out := pipeline.ValidatorOverride{Set: in.Set}
|
||||
if len(in.Validators) > 0 {
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user