Compare commits

91 Commits

Author SHA1 Message Date
916d9210fd Preserve structured repair settings across pipeline boundaries 2026-08-25 23:46:52 +00:00
3d3f16db4a Enable structural output repair by default 2026-08-25 20:04:56 +00:00
63c397d86a Resolve structured output repair configuration 2026-08-25 19:55:39 +00:00
9a92212632 Add structured output repair configuration 2026-08-25 19:47:42 +00:00
0ef8931697 Forward repair policy from D&D modules 2026-08-25 19:40:44 +00:00
e00cc45c6b Propagate structured repair requests 2026-08-25 19:39:04 +00:00
ab9b743df6 Add structured output repair support 2026-08-25 19:35:27 +00:00
3bd3c7ebf7 Classify PromptKit generation errors safely 2026-08-25 19:34:27 +00:00
75a3f51cee Support inherited PromptKit profiles 2026-08-25 19:31:00 +00:00
2be999ebd3 Verify PromptKit compatibility boundaries 2026-08-25 19:27:34 +00:00
32fe7c5b98 Upgrade PromptKit to v0.8.0 2026-08-25 19:25:57 +00:00
55247c47ab Plan the PromptKit 0.8 upgrade 2026-08-25 19:11:30 +00:00
5e5c69bf9d Document future validation and retry work 2026-08-25 14:29:00 +00:00
b4a81f8b09 Validate release versions before tagging 2026-08-25 02:21:46 +00:00
adfed22e1a Complete source release system verification 2026-08-25 02:05:17 +00:00
75b1e2f68b Document source release deployment 2026-08-25 02:03:11 +00:00
4473363d9f Document source release procedure 2026-08-25 02:01:52 +00:00
6eb45e0003 Add release tag validation workflow 2026-08-25 01:58:29 +00:00
0585ad76dc Add source release candidate checks 2026-08-25 01:57:10 +00:00
56145c3b7e Add build version reporting 2026-08-25 01:55:09 +00:00
a478fd86c5 Plan source-only releases 2026-08-25 01:48:16 +00:00
916532100d Add D&D consumer documentation 2026-08-09 20:57:28 +00:00
bef8ca263b Tighten evidence context source excerpts 2026-08-09 20:04:46 +00:00
f6d037b613 Document evidence context source unit excerpts 2026-08-09 19:47:27 +00:00
449b506804 Update evidence context output coverage 2026-08-09 19:45:21 +00:00
071a78ae22 Replace evidence context with source unit excerpts 2026-08-09 19:42:33 +00:00
ad1cba41c2 Fix OpenAI reconciliation schema 2026-08-09 18:49:29 +00:00
67338798aa Complete the semantic reconciliation roadmap 2026-08-09 18:30:32 +00:00
e95e2f2220 Finalize semantic reconciliation documentation 2026-08-09 17:05:17 +00:00
628b8d1800 Remove legacy reconciliation path 2026-08-09 16:57:44 +00:00
d24d4609b6 Migrate location reconciliation to shared engine 2026-08-09 16:51:29 +00:00
c7f79fb38e Migrate item reconciliation to shared engine 2026-08-09 16:46:43 +00:00
8c071800cf Migrate NPC reconciliation to shared engine 2026-08-09 16:38:43 +00:00
569e12c6f4 Add generic reconciliation plan application 2026-08-09 16:27:36 +00:00
5b6eb591b2 Add shared semantic reconciliation engine 2026-08-09 16:19:54 +00:00
b630384aa0 Add generic semantic reconciliation prompt assets 2026-08-09 16:08:53 +00:00
297d58f090 Add semantic reconciliation proposal validation 2026-08-09 16:00:01 +00:00
ee71dc4937 Add bounded semantic candidate preparation 2026-08-09 15:52:55 +00:00
65e5d65d14 Record semantic reconciliation architecture decision 2026-08-09 15:43:05 +00:00
b40b40aaf3 Plan semantic reconciliation improvements 2026-08-09 15:38:10 +00:00
f120be1cb4 Add proposed changes to shared LLM prompts 2026-08-09 09:18:36 -05:00
17673d74ea Archive the completed codebase audit 2026-08-09 13:24:12 +00:00
ef19a03cbf Reconcile remediation documentation 2026-08-09 02:45:22 +00:00
0546f6eb4f Simplify module cleanup paths 2026-08-09 02:40:25 +00:00
d28d1062e0 Reuse canonical item occurrence evidence 2026-08-09 02:36:53 +00:00
b3ebfcef37 Index enemy event duplicate identities 2026-08-09 02:31:28 +00:00
b70d9f77e3 Improve D&D registry normalization efficiency 2026-08-09 02:26:49 +00:00
2a75f40871 Bound D&D normalization diagnostics 2026-08-09 02:22:23 +00:00
a705ba74a1 Project spell aliases into extraction prompts 2026-08-09 02:17:15 +00:00
8d9c9e7c87 Align item occurrence evidence fields 2026-08-09 02:11:27 +00:00
0b5cc4f251 Require chunk-local extraction evidence 2026-08-09 02:08:11 +00:00
82ffe85f2d Enforce durable enemy event validation 2026-08-09 02:03:27 +00:00
b3644abc0e Enforce durable D&D evidence ranges 2026-08-09 01:54:43 +00:00
d653bf1b90 Share immutable in-memory filesystems 2026-08-09 01:47:44 +00:00
3e66127b94 Write durable outputs through confined file writer 2026-08-09 01:41:37 +00:00
5d086c13ca Cache compiled JSON schemas per validator 2026-08-09 01:38:42 +00:00
d36d4e7689 Avoid redundant decoded graph clones 2026-08-09 01:34:26 +00:00
1456aa51cc Index generated reference handoffs 2026-08-09 01:30:10 +00:00
ffc179c822 Centralize builder request cloning 2026-08-09 01:27:28 +00:00
8e669a1f14 Preserve terminal rejection warnings 2026-08-09 01:19:55 +00:00
14bfae216d Isolate typed validator candidates 2026-08-09 01:11:03 +00:00
5a58d87995 Require candidate decoders for artifact codecs 2026-08-09 01:03:50 +00:00
557809f364 Add candidate artifact codec decoding 2026-08-09 00:57:03 +00:00
ee600975f0 Prevent scheduler callbacks after cancellation 2026-08-09 00:52:31 +00:00
0d8017e23f Gate runner dispatches on cancellation 2026-08-09 00:50:02 +00:00
37b18edf3d Contain provider errors at the LLM adapter 2026-08-09 00:44:38 +00:00
cda7a61b47 Encode checkpoint and debug path identities 2026-08-09 00:41:28 +00:00
2ad9283148 Bound external reference reads and fingerprints 2026-08-09 00:33:33 +00:00
41a8a80dda Reject ambiguous configuration and reference bindings 2026-08-09 00:29:04 +00:00
90c7fa6381 Finalize codebase audit synthesis 2026-08-08 23:00:28 +00:00
e3839f8620 Audit combat and enemy event processing 2026-08-08 22:54:00 +00:00
0fc2f9ee01 Audit spell and scene processing 2026-08-08 22:42:49 +00:00
5d6305f21a Audit NPC item and location occurrences 2026-08-08 22:30:23 +00:00
551e4daea2 Audit NPC item and location registries 2026-08-08 22:18:29 +00:00
3589d33468 Audit shared D&D family conventions 2026-08-08 22:08:24 +00:00
a22c1a7f59 Audit generic and Seriatim modules 2026-08-08 21:59:06 +00:00
ad85d71b0f Audit LLM runtime and prompt assets 2026-08-08 21:47:51 +00:00
f3506240c2 Audit state persistence and file safety 2026-08-08 21:37:57 +00:00
70c199aa31 Audit runtime execution and concurrency 2026-08-08 21:27:37 +00:00
e2b82746ab Audit reference materialization and ordered handoffs 2026-08-08 21:16:30 +00:00
4235507f7b Audit pipeline composition and typed registries 2026-08-08 21:06:28 +00:00
b346670cc7 Audit configuration and CLI composition 2026-08-08 20:56:50 +00:00
7868c26be7 Establish codebase audit baseline 2026-08-08 20:50:15 +00:00
92e89076a2 Add an audit plan and staged audit sequence to identify opportunities for code quality improvement 2026-08-08 20:42:35 +00:00
d9b87347b8 Simplify contextual entity grounding 2026-08-08 15:47:41 +00:00
20397ef710 Document deterministic entity identity resolution 2026-08-08 15:08:57 +00:00
fc449863f2 Use contextual descriptors for entity reconciliation 2026-08-08 15:05:35 +00:00
51d62de1f3 Ground location occurrences with contextual selectors 2026-08-08 14:56:43 +00:00
fc76805075 Add contextual location grounding 2026-08-08 14:49:08 +00:00
8e680cf96e Ground item occurrences by canonical names 2026-08-08 14:43:04 +00:00
ece1bca460 Ground NPC occurrences by canonical names 2026-08-08 14:37:54 +00:00
307 changed files with 16455 additions and 3897 deletions

33
.woodpecker/release.yml Normal file
View File

@@ -0,0 +1,33 @@
when:
- event: tag
steps:
- name: validate-release
image: golang:1.25.5
commands:
- |
set -eu
version="$CI_COMMIT_TAG"
release_note="docs/releases/$version.md"
if ! printf '%s\n' "$version" | grep -E -x 'v(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)' >/dev/null; then
printf '%s\n' "invalid release tag: $version" >&2
exit 1
fi
if [ ! -s "$release_note" ]; then
printf '%s\n' "missing release note: $release_note" >&2
exit 1
fi
if ! grep -F -x "# Notarius $version" "$release_note" >/dev/null; then
printf '%s\n' "release note heading does not match $version" >&2
exit 1
fi
for heading in '## Summary' '## Compatibility' '## Upgrade' '## Changes'; do
if ! grep -F -x "$heading" "$release_note" >/dev/null; then
printf '%s\n' "release note is missing heading: $heading" >&2
exit 1
fi
done
./scripts/check-release-source.sh "$version"

View File

@@ -28,6 +28,20 @@ For the complete ordered D&D workflow, use
[its synthetic transcript](examples/dnd-complete-transcript.json). It
demonstrates all implemented D&D lanes and the supporting campaign references.
## Install A Source Release
Install a pinned source release with Go:
~~~
GOWORK=off go install \
gitea.maximumdirect.net/eric/notarius/cmd/notarius@<tag>
~~~
Replace `<tag>` with a stable release tag such as `vMAJOR.MINOR.PATCH`. The
installed command's diagnostic version is described in the [CLI
reference](docs/cli.md); maintainers preparing a release should follow [Source
Releases](docs/release.md).
## Documentation
- [CLI reference](docs/cli.md) — commands, flags, output streams, and exits.
@@ -39,6 +53,8 @@ demonstrates all implemented D&D lanes and the supporting campaign references.
artifact formats.
- [Subprocess consumer guide](docs/consumers/subprocess.md) — invoke Notarius
from an orchestrator and consume a published result.
- [Complete D&D consumer guide](docs/consumers/dnd-pipeline.md) — run the full
D&D pipeline as a subprocess and discover its structured artifacts.
- [Internal overview](docs/internal/overview.md) — implemented component map
for maintainers.
- [Developer guide](docs/development.md) — contributor orientation and

View File

@@ -42,4 +42,4 @@ output:
format: json
validation_mode: json_schema
schema_path: dnd_combat_turns_llm.v1.json
repair_attempts: 0
repair_attempts: 1

View File

@@ -50,4 +50,4 @@ output:
format: json
validation_mode: json_schema
schema_path: dnd_enemy_events_llm.v1.json
repair_attempts: 0
repair_attempts: 1

View File

@@ -1,24 +0,0 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "notarius.dnd.entity_reconcile.llm",
"type": "object",
"additionalProperties": false,
"required": ["duplicate_groups"],
"properties": {
"duplicate_groups": {
"type": "array",
"items": {
"type": "object",
"additionalProperties": false,
"required": ["members", "canonical"],
"properties": {
"members": {
"type": "array",
"items": {"type": "string"}
},
"canonical": {"type": "string"}
}
}
}
}
}

View File

@@ -3,10 +3,10 @@ in party possession established by the transcript. This is an occurrence history
not an inventory or ledger: do not calculate balances, resolve item identity
across records, or infer ownership that the transcript does not establish.
For every occurrence, copy the exact `item_id` and `name` pair from the supplied
item registry. Record a stated quantity as an integer and leave it null when the transcript
does not state one. Use a concise observed item name and preserve the stated
currency denomination.
For every occurrence, use the supplied canonical item `name`. Record a stated
quantity as an integer and leave it null when the transcript does not state
one. Preserve the stated currency denomination through the selected canonical
registry name.
Use `discovered` when the party learns of or encounters an item without
establishing possession. Use `acquired` when the party or a party member gains

View File

@@ -1,6 +1,6 @@
Use the supplied item registry only to ground each occurrence. Every record
must copy one registry item's exact `id` and exact `name`; do not invent,
rename, merge, or infer registry items. The registry is not transcript
evidence: cite only the current transcript chunk in `source_refs`.
must use one registry item's canonical `name`; do not invent, rename, merge,
or infer registry items. The registry is not transcript evidence: cite only the
current transcript chunk in `source_refs`.
{{ input "item_registry" }}

View File

@@ -42,4 +42,4 @@ output:
format: json
validation_mode: json_schema
schema_path: dnd_item_occurrences_llm.v1.json
repair_attempts: 0
repair_attempts: 1

View File

@@ -10,9 +10,8 @@
"items": {
"type": "object",
"additionalProperties": false,
"required": ["item_id", "name", "kind", "quantity", "from", "to", "source_refs"],
"required": ["name", "kind", "quantity", "from", "to", "source_refs"],
"properties": {
"item_id": {"type": "string"},
"name": {"type": "string"},
"kind": {"type": "string"},
"quantity": {"type": ["integer", "null"]},
@@ -23,10 +22,10 @@
"items": {
"type": "object",
"additionalProperties": false,
"required": ["start_segment", "end_segment"],
"required": ["start_unit_id", "end_unit_id"],
"properties": {
"start_segment": {"type": "integer"},
"end_segment": {"type": "integer"}
"start_unit_id": {"type": "integer"},
"end_unit_id": {"type": "integer"}
}
}
}

View File

@@ -37,4 +37,4 @@ output:
format: json
validation_mode: json_schema
schema_path: dnd_item_registry_llm.v1.json
repair_attempts: 0
repair_attempts: 1

View File

@@ -1,8 +1,9 @@
Use candidate names and cited transcript windows only to determine whether
candidates identify the same item type or unique designation. Do not treat
nearby evidence, similar objects, or a shared owner as sufficient. Keep
currency denominations, materially different item types, and uncertain aliases
separate. Do not infer an item property or uniqueness.
Determine whether candidates identify the same item type or unique designation
using their contextual labels and cited transcript windows. Do not treat nearby
evidence, similar objects, or a shared owner as sufficient.
Keep currency denominations and materially different item types separate. Keep
uncertain aliases separate. Do not infer an item property or uniqueness.
When selecting a canonical display name, choose one supplied candidate name
that is the clearest established designation.

View File

@@ -12,19 +12,19 @@ messages:
- role: system
content_file: ./sharedassets/common-dnd-system.md
- role: user
content_file: ./instructions.md
content_file: ./sharedassets/protocol.md
- role: user
content_file: ./sharedassets/common-dnd-entity-reconciliation.md
content_file: ./instructions.md
cache_control:
type: ephemeral
- role: user
content_file: ./candidates.md
content_file: ./sharedassets/candidates.md
- role: user
content_file: ./sharedassets/common-dnd-transcript-windows.md
content_file: ./sharedassets/transcript-windows.md
cache_control:
type: ephemeral
output:
format: json
validation_mode: json_schema
schema_path: dnd_entity_reconcile_llm.v1.json
repair_attempts: 0
schema_path: semantic_reconciliation_llm.v1.json
repair_attempts: 1

View File

@@ -20,6 +20,11 @@ location only when the chunk's context supports that coreference. It must not
create a registry location, and registry content or provenance must never
replace current-chunk evidence.
For every occurrence, return the exact selector from the location registry:
the canonical `name`, plus an empty `registry_refs` array for a unique name or
the complete ordered `registry_refs` array for a repeated name. Registry ranges
and context identify the location only; they are not occurrence evidence.
For overlapping support, visited outranks planned, recalled, and mentioned;
planned outranks recalled and mentioned; recalled outranks mentioned. A passage
may produce multiple records when it independently establishes separate facts,

View File

@@ -1,9 +1,11 @@
A normalized location registry is provided below for identity grounding. It may
be empty. Each record contains the exact location ID and canonical display name
to copy when the transcript establishes an occurrence of that place.
A contextual location registry is provided below for identity grounding. It may
be empty. Every record supplies a canonical display name. A name that appears
once is selected with that name and an empty `registry_refs` array. A repeated
name is selected only by copying both its name and its complete, ordered
`registry_refs` array exactly as supplied.
Registry content is context, not occurrence evidence. Do not derive an
occurrence or a source range from the registry, and do not infer a location
that is absent from it.
occurrence or `source_refs` range from the registry. Do not invent a location
or selector that is absent from it.
{{ input "location_registry" }}

View File

@@ -42,4 +42,4 @@ output:
format: json
validation_mode: json_schema
schema_path: dnd_location_occurrences_llm.v1.json
repair_attempts: 0
repair_attempts: 1

View File

@@ -10,10 +10,21 @@
"items": {
"type": "object",
"additionalProperties": false,
"required": ["location_id", "name", "kind", "source_refs"],
"required": ["name", "registry_refs", "kind", "source_refs"],
"properties": {
"location_id": {"type": "string"},
"name": {"type": "string"},
"registry_refs": {
"type": "array",
"items": {
"type": "object",
"additionalProperties": false,
"required": ["start_unit_id", "end_unit_id"],
"properties": {
"start_unit_id": {"type": "integer", "minimum": 1},
"end_unit_id": {"type": "integer", "minimum": 1}
}
}
},
"kind": {"enum": ["visited", "planned", "recalled", "mentioned"]},
"source_refs": {
"type": "array",

View File

@@ -37,4 +37,4 @@ output:
format: json
validation_mode: json_schema
schema_path: dnd_location_registry_llm.v1.json
repair_attempts: 0
repair_attempts: 1

View File

@@ -1,2 +0,0 @@
Location candidates:
{{ input "candidates" }}

View File

@@ -1,6 +1,8 @@
Use candidate names and their cited transcript windows to determine whether
candidates identify the same physical place. Do not treat matching names,
nearby evidence, nested places, or generic labels as sufficient. Keep parent
and child places, similarly named places, and uncertain aliases separate.
Determine whether candidates identify the same physical place using their
contextual labels and cited transcript windows. Do not treat matching names,
nearby evidence, nested places, or generic labels as sufficient.
Keep parent and child places separate, as well as similarly named places and
uncertain aliases.
When selecting a canonical display name, prefer the clearest established name.

View File

@@ -12,19 +12,19 @@ messages:
- role: system
content_file: ./sharedassets/common-dnd-system.md
- role: user
content_file: ./instructions.md
content_file: ./sharedassets/protocol.md
- role: user
content_file: ./sharedassets/common-dnd-entity-reconciliation.md
content_file: ./instructions.md
cache_control:
type: ephemeral
- role: user
content_file: ./candidates.md
content_file: ./sharedassets/candidates.md
- role: user
content_file: ./sharedassets/common-dnd-transcript-windows.md
content_file: ./sharedassets/transcript-windows.md
cache_control:
type: ephemeral
output:
format: json
validation_mode: json_schema
schema_path: dnd_entity_reconcile_llm.v1.json
repair_attempts: 0
schema_path: semantic_reconciliation_llm.v1.json
repair_attempts: 1

View File

@@ -1,8 +1,8 @@
Extract Dungeons & Dragons NPC occurrences from the supplied
transcript. Include an occurrence only when the transcript establishes one
supplied NPC, one occurrence kind, and a coherent passage supporting both.
Use the exact `npc_id` and matching `name` pair from the supplied NPC registry;
never invent an ID or substitute a similar name.
Use the supplied canonical NPC `name`; never invent or substitute a similar
name. Cite current-transcript evidence for every occurrence.
Do not summarize, infer relationships, sentiment, factions, motives, aliases,
or persistent state. Do not identify player characters, anonymous groups, or

View File

@@ -42,4 +42,4 @@ output:
format: json
validation_mode: json_schema
schema_path: dnd_npc_occurrences_llm.v1.json
repair_attempts: 0
repair_attempts: 1

View File

@@ -10,11 +10,8 @@
"items": {
"type": "object",
"additionalProperties": false,
"required": ["npc_id", "name", "kind", "source_refs"],
"required": ["name", "kind", "source_refs"],
"properties": {
"npc_id": {
"type": "string"
},
"name": {
"type": "string"
},

View File

@@ -37,4 +37,4 @@ output:
format: json
validation_mode: json_schema
schema_path: dnd_npc_registry_llm.v1.json
repair_attempts: 0
repair_attempts: 1

View File

@@ -1,3 +0,0 @@
NPC candidates for identity comparison:
{{ input "candidates" }}

View File

@@ -1,6 +1,7 @@
Use candidate aliases and their cited transcript windows to determine whether
candidates refer to the same individual. Preserve distinct individuals even
when their names are similar.
Determine whether candidates refer to the same individual using their
contextual labels and cited transcript windows. Preserve distinct individuals
even when their names are similar or their contextual descriptions are
identical.
When selecting a canonical display name, prefer a complete, stable proper name
over an abbreviation. Prefer an unadorned proper name over that name plus a

View File

@@ -12,19 +12,19 @@ messages:
- role: system
content_file: ./sharedassets/common-dnd-system.md
- role: user
content_file: ./instructions.md
content_file: ./sharedassets/protocol.md
- role: user
content_file: ./sharedassets/common-dnd-entity-reconciliation.md
content_file: ./instructions.md
cache_control:
type: ephemeral
- role: user
content_file: ./candidates.md
content_file: ./sharedassets/candidates.md
- role: user
content_file: ./sharedassets/common-dnd-transcript-windows.md
content_file: ./sharedassets/transcript-windows.md
cache_control:
type: ephemeral
output:
format: json
validation_mode: json_schema
schema_path: dnd_entity_reconcile_llm.v1.json
repair_attempts: 0
schema_path: semantic_reconciliation_llm.v1.json
repair_attempts: 1

View File

@@ -35,4 +35,4 @@ output:
format: json
validation_mode: json_schema
schema_path: dnd_scene_descriptions_llm.v1.json
repair_attempts: 0
repair_attempts: 1

View File

@@ -31,4 +31,4 @@ output:
format: json
validation_mode: json_schema
schema_path: dnd_scenes_llm.v1.json
repair_attempts: 0
repair_attempts: 1

View File

@@ -1,6 +0,0 @@
Identify only well-supported duplicate groups among the supplied candidates.
Candidate keys are opaque identifiers. Copy each selected key exactly. A group
must contain at least two supplied keys, and its `canonical` key must be one of
its members. Do not create keys, records, names, source references, evidence,
or replacement values. Omit any uncertain or unsafe group.

View File

@@ -1,7 +1,6 @@
Transcript units are the only evidence for extracted events and factual claims.
Every reported factual claim must be supported by cited transcript units. Use
integer `start_unit_id` and `end_unit_id` values from the transcript. Omit
`source_id`; Notarius assigns the current source identity.
integer `start_unit_id` and `end_unit_id` values from the transcript.
When supporting evidence is non-contiguous, use multiple narrow ranges rather
than a broad range that bridges unrelated conversation.

View File

@@ -1,7 +1,5 @@
You process Dungeons & Dragons gameplay transcripts.
Rely only on the supplied inputs. They may contain transcription errors,
repeated lines, incomplete sentences, and misheard proper nouns.
As input, you will receive one or more portions of a transcript. The transcript may contain transcription errors, repeated lines, incomplete sentences, and misheard proper nouns.
Return exactly one JSON object that conforms to the configured response schema,
with no explanatory prose.
Return exactly one JSON object that conforms to the configured response schema, with no explanatory prose.

View File

@@ -1,5 +1,3 @@
One extraction chunk from a Dungeons & Dragons gameplay transcript is provided
below. Report and infer only what is within this chunk. Its unit IDs retain
their source-wide meaning.
One extraction chunk from a Dungeons & Dragons gameplay transcript is provided below. Report and infer only what is within this chunk. Its unit IDs retain their source-wide meaning.
{{ input "transcript" }}

View File

@@ -1,4 +1,3 @@
The complete ordered transcript of this Dungeons & Dragons gameplay session is
provided below. It may contain multiple scenes.
The complete ordered transcript of this Dungeons & Dragons gameplay session is provided below.
{{ input "transcript" }}

View File

@@ -1,6 +0,0 @@
Selected Dungeons & Dragons gameplay transcript evidence windows are provided
below. They may be incomplete, non-contiguous, or overlapping. Use them to
evaluate candidate identity, but do not treat absence outside these windows as
evidence.
{{ input "transcript" }}

View File

@@ -47,4 +47,4 @@ output:
format: json
validation_mode: json_schema
schema_path: dnd_spells_llm.v1.json
repair_attempts: 0
repair_attempts: 1

View File

@@ -1,6 +1,6 @@
The canonical spell-name catalog for this extraction is provided below as JSON.
Return spell names using the catalog's canonical spelling exactly. Aliases and
other campaign reference material are not part of this catalog input and must
not be copied into the output as spell names.
The spell catalog for this extraction is provided below as JSON. Each entry
lists a `canonical_name` and its recognized `aliases`. If the transcript uses
an alias, select that entry's `canonical_name`. Return spell names using the
canonical spelling exactly; never return an alias as a spell name.
{{ input "spell_catalog" }}

View File

@@ -1,2 +1,3 @@
Item candidates:
Candidate material:
{{ input "candidates" }}

View File

@@ -0,0 +1,5 @@
Identify only high-confidence duplicate entities among the supplied candidates.
Preserve distinct entities even when their names are similar. Treat contextual descriptions and transcript evidence as supporting material, not as permission to merge ambiguous records.
When several records are duplicates, choose as canonical the candidate with the clearest stable identity. Prefer a complete proper name over an abbreviation, and prefer an unadorned proper name over one with incidental descriptors unless the evidence establishes those descriptors as part of the name. A longer name is not inherently more canonical.

View File

@@ -0,0 +1,27 @@
id: generic.semantic_reconciliation
version: "v1"
inputs:
- name: candidates
required: true
content_type: application/json
- name: transcript
required: true
content_type: application/json
messages:
- role: system
content_file: ./system.md
- role: user
content_file: ./protocol.md
- role: user
content_file: ./instructions.md
cache_control:
type: ephemeral
- role: user
content_file: ./candidates.md
- role: user
content_file: ./transcript-windows.md
output:
format: json
validation_mode: json_schema
schema_path: semantic_reconciliation_llm.v1.json
repair_attempts: 1

View File

@@ -0,0 +1,7 @@
Use only the positive integer `candidate_id` values supplied in the candidate material.
Return a duplicate group only when the evidence supports that every selected candidate describes the same underlying entity. Each group must contain at least two distinct candidate IDs, and its `canonical_candidate_id` must be one of those IDs. A candidate may appear in at most one group.
Omit uncertain matches and candidates that should remain distinct. Do not invent candidates or infer an ID from list position. An empty `duplicate_groups` array is valid.
The response must conform exactly to the selected JSON schema. Return IDs only: do not copy candidate names, evidence, transcript text, source identifiers, or source ranges into the response.

View File

@@ -0,0 +1,2 @@
You reconcile structured records that may describe the same underlying entity.
Follow the supplied protocol and return only the requested structured result.

View File

@@ -0,0 +1,3 @@
Transcript evidence windows:
{{ input "transcript" }}

View File

@@ -0,0 +1,32 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "notarius.generic.semantic_reconciliation.llm",
"title": "notarius_semantic_reconciliation_llm_v1",
"type": "object",
"additionalProperties": false,
"required": ["duplicate_groups"],
"properties": {
"duplicate_groups": {
"type": "array",
"items": {
"type": "object",
"additionalProperties": false,
"required": ["candidate_ids", "canonical_candidate_id"],
"properties": {
"candidate_ids": {
"type": "array",
"minItems": 2,
"items": {
"type": "integer",
"minimum": 1
}
},
"canonical_candidate_id": {
"type": "integer",
"minimum": 1
}
}
}
}
}
}

View File

@@ -0,0 +1,64 @@
# ADR-0012: Resolve opaque entity identifiers deterministically
**Status:** Accepted
**Date:** 2026-08-08
## Context
Entity IDs in durable Notarius artifacts are application-owned, deterministic
identifiers. They are useful to artifact consumers, but their hash-based form
does not help a model distinguish entities and would make the model reproduce
an opaque implementation detail. A plain name is likewise insufficient where
multiple supplied records share that name.
The LLM boundary must preserve the typed artifact and durable-schema ownership
of [ADR-0003](0003-typed-interfaces-with-two-zone-data-model.md) and the distinction
between disambiguating references and source evidence in
[ADR-0009](0009-minimal-evidence-grounded-extraction-artifacts.md).
## Decision
Callers present a model with semantic selections: a canonical name when it is
unique in the request, or a contextual descriptor containing the name and
source coordinates when that context is needed to distinguish supplied
records. The model returns only those supplied selections. The caller resolves
each accepted selection against the request-local supplied records and attaches
the opaque application ID deterministically.
Source coordinates are permitted in a selection solely as identity context.
They neither establish an occurrence fact nor replace that occurrence's
current-transcript evidence. A selector must resolve exactly; unknown,
ambiguous, partial, reordered, or otherwise unsafe selections are not mapped.
Where an operation requires a complete grounded artifact, that failure rejects
the complete artifact rather than accepting a partially mapped result.
An explicitly scoped request-local short label is permitted only when a
contextual descriptor would be impractical and the caller can deterministically
map the label within that one request. Such a label is not a durable ID, must
not escape the request boundary, and requires a concrete justification in its
own module contract.
## Alternatives considered
- Ask the model to return durable IDs. This exposes opaque implementation
state, does not improve semantic disambiguation, and makes model output
depend on hash formatting.
- Select by name alone. This cannot safely distinguish same-name records.
- Make request-local labels durable identifiers. This would turn prompt
presentation into a public identity contract and create avoidable migration
pressure.
- Let the model invent identifiers or resolve ambiguity. This makes identity
assignment non-deterministic and weakens validation.
## Consequences
Durable integration contracts retain their exact ID/name pairs while models
operate on readable contextual selections. Calling modules must own selector
construction, exact resolution, ambiguity handling, and conversion into their
durable artifact type; PromptKit and its adapter remain transport-only.
Some ambiguous or invalid proposals are deliberately omitted, retried, or
rejected according to the caller's existing failure policy. Internal candidate
keys may support deterministic request-local mapping, but they are not
model-visible selectors or durable data. This adds local validation work while
keeping identity assignment auditable and stable.

View File

@@ -0,0 +1,91 @@
# ADR-0013: Use request-local candidate handles for semantic reconciliation
**Status:** Accepted
**Date:** 2026-08-09
## Context
Several typed normalize stage modules need semantic reconciliation after
deterministic preprocessing: a model can judge whether source-backed candidates
refer to the same underlying entity, while application code remains responsible
for constructing the normalized artifact. Requiring the model to reproduce a
candidate's full contextual selector makes the response larger and introduces
avoidable formatting, ordering, and transcription failure modes.
Reconciliation must preserve the exact typed artifact boundary established by
[ADR-0003](0003-typed-interfaces-with-two-zone-data-model.md), the domain-neutral
framework and concrete-domain dependency direction established by
[ADR-0004](0004-package-modules-by-domain.md), and the distinction in
[ADR-0009](0009-minimal-evidence-grounded-extraction-artifacts.md) between source
evidence and auxiliary identity context. It also needs a concrete, narrowly
scoped application of the request-local-label exception allowed by
[ADR-0012](0012-resolve-opaque-entity-identifiers-deterministically.md).
## Decision
Semantic reconciliation will be a domain-neutral framework mechanism used by
typed normalize stage modules. A consuming artifact family will retain
ownership of its typed records, identity rules, consolidation policy, durable
IDs, and domain warnings; the framework mechanism will not infer those rules
from arbitrary data.
For each reconciliation request, deterministic code will assign every eligible
model-visible candidate a contiguous, one-based integer handle. The model may
receive the candidate's contextual label, source references, and bounded source
context needed to judge identity, but its structured response will identify
candidates only by those supplied handles. A handle is local to one request,
does not represent entity identity, and must never enter a durable artifact or
be used to derive a durable ID.
The model will propose duplicate groups and select one supplied member of each
group as canonical. Deterministic code will resolve the handles through the
retained request mapping, validate the complete proposal, discard unsafe
groups, and apply only validated groups through typed domain-owned policy. The
model will not synthesize replacement records or directly mutate an artifact.
Every reconciliation prompt will combine a mandatory framework-owned protocol
and safety policy with an explicitly selected semantic policy. The semantic
policy may be the conservative generic policy or a domain-owned policy, but it
cannot replace the shared response protocol or deterministic safety boundary.
## Alternatives considered
- Return durable application IDs. Opaque IDs do not help semantic judgment,
expose application identity mechanics, and make model output reproduce data
that deterministic code already owns.
- Return names alone or copied contextual selectors. Names can be ambiguous,
while reproducing labels and source ranges adds response complexity and
creates mismatches without adding semantic information. Request-local
handles preserve exact selection without either failure mode.
- Ask the model to return synthesized canonical replacement records. This
would transfer typed artifact construction, provenance consolidation, and
durable identity policy to a probabilistic boundary.
- Reconcile reflection-discovered fields or arbitrary JSON. This would weaken
the typed artifact contract and move domain semantics into generic code.
- Hide reconciliation inside extraction or another stage. This would obscure
stage ownership and create cross-stage behavior outside the fixed pipeline;
reconciliation remains explicit normalize-stage behavior.
- Let each domain replace the complete prompt protocol. This would duplicate
safety mechanics and allow domain policy to bypass the common response and
validation contract.
## Consequences
Model responses become smaller and easier to validate, while deterministic
application code retains authority over identity, provenance, ordering, and
typed artifact construction. The framework requires a request-local mapping,
bounded context preparation, a private integer response contract, proposal
assessment, and shared prompt assets. Each consuming artifact family still
requires a typed adapter for its irreducibly domain-specific rules.
Request-local handles are deliberately unsuitable for persistence, logging as
entity identity, checkpoint contracts, or cross-request correlation. Changes
to shared protocol and policy assets must participate in the normal prompt,
schema, and checkpoint fingerprint mechanisms.
Acceptance of this decision does not imply that the shared mechanism or its
consumer migrations are implemented. The
[feature roadmap](../roadmap/semantic-reconciliation.md) owns target behavior
and status, and the
[implementation plan](../roadmap/implementation.md) owns delivery sequence
until the work is complete.

View File

@@ -10,6 +10,7 @@ defined in [Operations](operations.md).
~~~
notarius help
notarius --version
notarius run <pipeline-id> --input path/to/source.json [--json] [flags]
notarius config validate [--config path/to/config.yml] [--pipeline pipeline-id] [--only lane-a,lane-b]
notarius pipelines list [--config path/to/config.yml] [--json]
@@ -18,6 +19,17 @@ notarius pipelines list [--config path/to/config.yml] [--json]
Running Notarius without arguments, or with **help**, **--help**, or **-h**,
writes the command summary to standard output and exits with status 0.
`notarius --version` is valid only as the sole root argument. It writes exactly
`notarius <version>` followed by a newline to standard output and exits with
status 0. A tagged `go install` build can report its main-module stable tag,
and controlled builds can inject a stable tag at link time through
`gitea.maximumdirect.net/eric/notarius/internal/buildinfo.Override`; an ordinary
unversioned checkout reports `development`. Invalid injected version content is
a runtime error with exit status 1, while extra `--version` arguments are a
syntax error with exit status 2. This diagnostic does not replace the
[run-result](integrations/run-result.md) or artifact contracts for downstream
compatibility decisions.
## run
~~~

View File

@@ -132,7 +132,11 @@ model: example-model
Keep credentials out of the local-backend object. A PromptKit profile may name
its credential environment variable through `api_key_env`; set that variable
only in the run environment. PromptKit owns the
[pinned profile-file format](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0/docs/formats.md).
[pinned profile-file format](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.8.0/docs/formats.md),
including `base_profile` inheritance. Notarius passes profiles through without
merging them. Filesystem profiles cannot express PromptKit's in-memory
`APIKeyRequired` setting; an unset `api_key_env` is optional and may reach the
provider without authorization.
The [PromptKit upstream boundary](integrations/pkg-promptkit.md) identifies the
supported package API, and [Operations](operations.md#operational-limits)
describes the effective concurrency layers.
@@ -218,6 +222,7 @@ pipelines:
| Field | Type | Default | Rules |
| --- | --- | --- | --- |
| **llm_profile** | string | none | Optional non-empty default PromptKit profile ID for selected LLM-backed bindings and validators. An explicitly present blank value is invalid. |
| **structured_output_repair_attempts** | integer | prompt-owned (1 in maintained production prompts) | Optional structural-repair limit from 0 through 3 for selected LLM-backed bindings and validators. Omission leaves the prompt's declared policy in control; explicit 0 disables structural repair at that scope. |
| **input** | module binding | none | Required. |
| **chunk** | module binding | **generic** | Optional. |
| **output** | module binding | **json** | Optional. |
@@ -237,6 +242,15 @@ run-level **--llm-profile** value first, then the binding's **llm_profile**,
then the pipeline's **llm_profile**, and finally the PromptKit default.
Deterministic bindings do not receive these defaults or run overrides.
Structural output repair is resolved after module, validator, and `--only` lane
selection. An object's **structured_output_repair_attempts** value takes
precedence over the pipeline value; otherwise, an LLM-backed binding or
validator inherits the pipeline value. If both are omitted, PromptKit uses the
prompt's declared repair policy. The value must be an integer from 0 through 3;
explicit `null` and non-integer values are invalid. An explicit value on a
deterministic binding or validator is invalid, while a pipeline value simply
does not apply to deterministic selections.
A lane has these fields:
| Field | Type | Default | Rules |
@@ -274,6 +288,7 @@ extract:
| --- | --- | --- | --- |
| **module** | string | none | Required for an object binding. Must be a registered compatible key. |
| **llm_profile** | string | none | Optional non-empty PromptKit profile ID for an LLM-backed binding. It overrides the pipeline default unless the run supplies **--llm-profile**. |
| **structured_output_repair_attempts** | integer | pipeline or prompt-owned (1 in maintained production prompts) | Optional structural-repair limit from 0 through 3 for an LLM-backed binding. It overrides the pipeline value; explicit 0 disables structural repair. |
| **retries** | integer | 0 | Non-negative additional attempts for chunk, extract, merge, and normalize bindings. |
| **options** | object | none | Must satisfy the selected module. |
| **references** | map | none | Valid only on chunk, extract, merge, and normalize bindings. |
@@ -281,10 +296,11 @@ extract:
Omitting **validators** uses the registered chain. **validators: []** selects
an empty chain; a non-empty list replaces the chain in the listed order.
Validator bindings accept only **module**, **llm_profile**, and **options**.
They reject **references**, **retries**, and nested **validators**. Deterministic
validators reject an explicit **llm_profile**. Deterministic module bindings
also reject an explicit **llm_profile**.
Validator bindings accept only **module**, **llm_profile**,
**structured_output_repair_attempts**, and **options**. They reject
**references**, **retries**, and nested **validators**. Deterministic validators
reject explicit **llm_profile** and **structured_output_repair_attempts**.
Deterministic module bindings also reject those explicit fields.
The **json** output module accepts optional **include_chunk_map** and
**evidence_context** settings:
@@ -319,7 +335,8 @@ Unknown outer or nested option fields are rejected, as are incompatible YAML
types. The allowlist remains valid when a run uses lane filtering: a configured
lane that is not active for that invocation simply contributes no evidence.
Evidence publication is opt-in because it can persist source text and metadata.
Its payload contract is [Published Evidence Context](integrations/evidence-context.md).
When enabled, it publishes the selected source-unit excerpt defined by the
[Published Evidence Context contract](integrations/evidence-context.md).
## References And Ordered Handoffs

View File

@@ -0,0 +1,202 @@
# Consuming The Complete D&D Pipeline
Use this workflow when an orchestrator runs the maintained complete D&D
pipeline and consumes its structured JSON artifacts. The generic
[subprocess consumer guide](subprocess.md) owns process-level responsibilities;
this guide connects that workflow to the complete D&D configuration, its
Seriatim input, and its artifact inventory.
The [CLI reference](../cli.md), [configuration reference](../config.md),
[run-result receipt](../integrations/run-result.md), and
[published JSON output contract](../integrations/json-output.md) remain the
canonical definitions of those public interfaces.
## Prepare And Validate The Deployment
Start from the maintained
[complete D&D configuration](../../examples/dnd-complete.config.yml). It uses
the `dnd-session` pipeline and demonstrates every implemented D&D lane, ordered
artifact handoffs, campaign references, chunk-map publication, and evidence
context.
A deployment must provide its own PromptKit profile and campaign reference
files. Use absolute paths for service and subprocess deployments. In
particular, observe these different resolution rules:
- reference paths in YAML are resolved relative to the Notarius configuration
file; and
- `promptkit.profile_file` is resolved relative to the Notarius process working
directory.
Do not copy the repository example's relative profile path into a deployment
without also controlling that working directory. The complete path and profile
rules are defined in [Configuration](../config.md).
Preflight the deployed configuration before processing sessions and whenever
it changes:
```sh
notarius config validate \
--config /absolute/path/to/notarius.yml \
--pipeline dnd-session
```
Provide credentials through the environment or the documented configuration
mechanism. Do not put credentials in command arguments, generated
configuration, or logs.
## Supply The Transcript
The complete pipeline consumes a Seriatim JSON document. The
[Seriatim input contract](../integrations/seriatim.md) defines its required
metadata, segments, and validation rules. Preserve segment IDs: D&D artifact
citations use those segment IDs as source-unit ranges.
When the caller maintains several transcript tiers, use the final trimmed JSON
transcript so extraction operates on the same session content presented to
later consumers. For example, Narratio identifies this implemented artifact as
`narratio.transcript.final_trimmed` and normally stores it at
`transcripts/final.trimmed.json`.
Notarius generates a stable prompt session from the resolved input module and
the exact input bytes. An ordinary orchestrator should not pass `--session-id`.
Use that override only when intentionally changing the routing relationship
between invocations; it is not a credential or output identity.
## Run Notarius
Invoke the pipeline with explicit absolute paths and request its
machine-readable receipt:
```sh
notarius run dnd-session \
--config /absolute/path/to/notarius.yml \
--input /absolute/path/to/transcripts/final.trimmed.json \
--output-dir /absolute/path/to/notarius-output \
--json
```
The caller should:
- capture stdout and stderr separately;
- propagate cancellation and impose an operator-appropriate timeout;
- wait for process completion before interpreting stdout; and
- retain stderr for diagnosis without copying secrets or transcript content
into other logs.
Only exit status 0 permits decoding stdout as a receipt. Ignore stdout after a
nonzero exit because a failed receipt write can leave partial bytes. The
[CLI reference](../cli.md#output-streams-and-exit-statuses) defines the complete
stream and exit-status contract.
## Discover The Published Bundle
Decode the successful stdout document as a supported run-result schema. For
the current contract, `schema_version` is `notarius.run-result.v1`. Tolerate
unknown fields allowed by that version, but reject an unsupported schema
version.
Use the receipt's absolute `output_directory` as the exact run-specific bundle
root. Do not scan the output root for its newest directory, guess a run ID, or
construct a bundle path. Resolve `index_file` beneath `output_directory` and
reject an absolute logical path or any result that escapes the bundle root.
Read `index.json` and locate each requested lane in `output_files` by its exact
`lane_id`. Do not guess a lane filename. Before decoding a payload:
1. resolve its descriptor's relative `file` beneath the bundle root with the
same confinement check;
2. verify the descriptor's media type and schema identity against the linked
artifact contract; and
3. decode the payload according to that contract.
The [published JSON output contract](../integrations/json-output.md) defines
the index and bundle layout. Treat all paths obtained from a decoded external
document as untrusted until confined to their documented root.
## Complete Artifact Inventory
When every configured lane is accepted, the complete example publishes these
lane artifacts:
| Lane ID | Purpose | Canonical contract |
| --- | --- | --- |
| `item-registry` | Canonical registry of encountered items and currency. | [Item registry](../integrations/dnd-item-registry-artifacts.md) |
| `npc-registry` | Canonical registry of named NPCs. | [NPC registry](../integrations/dnd-npc-registry-artifacts.md) |
| `location-registry` | Canonical registry of named locations. | [Location registry](../integrations/dnd-location-registry-artifacts.md) |
| `scene-descriptions` | Classification, title, and summary for each scene. | [Scene descriptions](../integrations/dnd-scene-description-artifacts.md) |
| `item-occurrences` | Source-grounded item discovery, acquisition, use, transfer, and loss events. | [Item occurrences](../integrations/dnd-item-occurrence-artifacts.md) |
| `spells` | Source-grounded spell casts and casters. | [Spell casts](../integrations/dnd-spell-artifacts.md) |
| `combat-turns` | Source-grounded combat turn participation. | [Combat turns](../integrations/dnd-combat-turn-artifacts.md) |
| `npc-occurrences` | Source-grounded NPC interaction occurrences. | [NPC occurrences](../integrations/dnd-npc-occurrence-artifacts.md) |
| `location-occurrences` | Source-grounded location occurrences. | [Location occurrences](../integrations/dnd-location-occurrence-artifacts.md) |
| `enemy-events` | Source-grounded enemy combat events. | [Enemy events](../integrations/dnd-enemy-event-artifacts.md) |
The JSON encoder always publishes these bundle-management files:
| File | Purpose |
| --- | --- |
| `index.json` | Discovery document for lane and pipeline-wide artifacts. |
| `manifest.json` | Run provenance and result summaries. |
| `rejected.json` | Rejected pipeline outputs. |
| `warnings.json` | Accepted-output and run warnings. |
The complete configuration also requests two pipeline-wide artifacts:
- [`chunk-map.json`](../integrations/chunk-map.md), the accepted chunk plan and
chunk metadata; and
- [`evidence-context.json`](../integrations/evidence-context.md), a reading
excerpt containing the union of selected cited source units and the
configured surrounding window.
Discover both from their top-level `index.json` descriptors rather than
treating them as lanes. Evidence context is convenient reading material, not
authoritative provenance; citations in the normalized lane payloads remain the
evidence contract.
Every optional or lane file is published only when its corresponding artifact
is available. A successful process does not guarantee that all configured
lanes were accepted.
## Decide What Counts As Consumer Success
Exit status 0 means Notarius completed the pipeline and published its result
bundle. The receipt or bundle may still report warnings, rejected outputs, or
missing lane descriptors. A downstream consumer must define its own required
artifact set explicitly.
A caller that claims to consume the complete D&D workflow should normally
require all ten lane IDs in the table and verify each descriptor's expected
contract. If any required lane is missing, rejected, or incompatible, fail the
caller's extraction step while retaining the Notarius bundle for diagnosis. A
consumer that needs only a subset may define and document a narrower policy.
Keep the successful receipt with the complete published bundle. Retain
`manifest.json`, `rejected.json`, `warnings.json`, and captured process logs as
required by the caller's provenance, diagnosis, and retention policies. Avoid
selectively copying payload files without also preserving enough index and
manifest information to identify their originating run and contracts.
The transcript, lane artifacts, evidence context, manifest, debug data, and
logs can all contain private campaign information. Apply the same access,
publication, and retention controls used for the source transcript.
## Consumer Checklist
- Validate the deployed Notarius configuration and `dnd-session` pipeline.
- Pass the final trimmed Seriatim JSON transcript with stable segment IDs.
- Use absolute configuration, input, output-root, profile, and reference paths
in service deployments.
- Capture stdout and stderr separately and enforce cancellation and timeout.
- Parse stdout only after exit status 0.
- Accept only supported receipt, index, and artifact schema versions while
tolerating permitted unknown fields.
- Use the receipt's `output_directory`; never guess the run directory.
- Confine `index_file` and every descriptor path to the published bundle root.
- Discover lanes by `lane_id` and verify descriptor compatibility before
decoding payloads.
- Enforce an explicit required-lane policy and inspect rejections and warnings.
- Preserve the receipt and sufficient bundle provenance for every retained
artifact.
- Protect all transcript-derived files and diagnostic streams as sensitive
campaign data.

View File

@@ -6,6 +6,10 @@ statuses, while the [run-result receipt](../integrations/run-result.md) and
[Published JSON Output contract](../integrations/json-output.md) own the
durable result formats.
For the maintained complete D&D workflow, including its transcript input,
configured lane inventory, and downstream acceptance checklist, see
[Consuming The Complete D&D Pipeline](dnd-pipeline.md).
## Run And Check The Process
Optionally preflight a selected configuration and pipeline before work starts:
@@ -55,10 +59,10 @@ contract. The JSON bundle contract links to the available lane contracts.
If `index.json` has an `evidence_context` descriptor, treat it as a
pipeline-wide artifact rather than a lane entry. Verify its six descriptor
fields before decoding the linked file according to the [Published Evidence
Context contract](../integrations/evidence-context.md). Use each
`evidence_refs` entry as the citation to source material. Its surrounding
context range and included units explain the citation, but do not widen or
replace the cited source reference.
Context contract](../integrations/evidence-context.md). Decode its top-level
source-unit array as a reading excerpt. Obtain authoritative citations and lane
provenance from the normalized lane artifacts; the excerpt has neither and its
nearby units do not widen a lane artifact's cited source reference.
A zero exit status may still report rejected outputs, warnings, or absent
lanes. The caller decides which lane IDs are required for its own work and
@@ -73,5 +77,5 @@ them. Treat the input, output bundle, cache, debug bundle, and captured process
logs as potentially sensitive data. Apply the caller's access controls and
retention policy, and avoid copying secrets into arguments, logs, or
provenance records. An evidence-context artifact contains source-unit text and
metadata, and selected lanes can cover most of an input; preserve and share it
only when that source content is authorized for the recipient.
metadata and can cover most of an input; preserve and share it only when that
source content is authorized for the recipient.

View File

@@ -18,13 +18,14 @@ implemented component map.
| Any documentation addition or revision | [Documentation Policy](policy/documentation.md) | It defines canonical homes, audiences, current-behavior rules, and maintenance requirements. |
| Adding, changing, reviewing, or deleting tests | [Testing Policy](policy/testing.md) | It defines risk-based sufficiency, durable test boundaries, test-double guidance, and criteria for retaining tests. |
| CLI composition or command behavior | [CLI Internals](internal/cli.md) and [CLI Reference](cli.md) | The internal guide owns composition and command flow; the reference owns public syntax. |
| Building a subprocess caller or changing its result protocol | [Subprocess Consumer Guide](consumers/subprocess.md), [Run Result Receipt](integrations/run-result.md), and [CLI Internals](internal/cli.md) | These separate caller workflow, durable receipt contract, and CLI implementation behavior. |
| Building a subprocess caller or changing its result protocol | [Subprocess Consumer Guide](consumers/subprocess.md), [Complete D&D Consumer Guide](consumers/dnd-pipeline.md), [Run Result Receipt](integrations/run-result.md), and [CLI Internals](internal/cli.md) | These separate generic caller workflow, the complete D&D workflow, the durable receipt contract, and CLI implementation behavior. |
| Configuration loading, resolution, or user-visible configuration behavior | [Configuration Internals](internal/configuration.md) and [Configuration](config.md) | The internal guide owns loading and resolution mechanics; the reference owns the configuration contract. |
| Pipeline resolution or execution | [Pipeline Internals](internal/pipeline.md) | It documents profiles, references, validation, retries, checkpoints, and runner behavior. |
| Production modules or validators | [Module Internals](internal/modules.md), [D&D Module Internals](internal/dnd.md), and [D&D integration contracts](integrations/) | The generic guide owns extension mechanics, the D&D guide owns shared family conventions, and the contracts own durable output shapes. |
| LLM clients, prompts, schemas, profiles, or scheduling | [LLM Runtime](internal/llm.md) | It documents the transport boundary and PromptKit integration. |
| Output, cache, resume, or debug artifacts | [Run State Internals](internal/state.md), [Operations](operations.md), and [Configuration](config.md) | These separate implementation details, operator behavior, and configuration contracts. |
| External input formats, artifact schemas, or durable output files | [Integration Contracts](integrations/) | Integration documents define external and durable data contracts. |
| Release preparation, tagging, publication, or verification | [Source Releases](release.md) and [Documentation Policy](policy/documentation.md) | The release procedure owns maintainer guards and immutable-tag recovery; the policy assigns release-note ownership. |
| Proposed or unimplemented behavior | [Roadmap](roadmap/) | Future work belongs only in roadmap documentation until implemented. |
For an existing subsystem, also inspect its focused tests and the package-local

View File

@@ -24,12 +24,13 @@ An incompatible shape change requires a new schema version.
Both extraction and normalization require an `item_registry` reference bound to
an earlier normalized `dnd/item-registry` artifact. The registry is immutable
for an operation and contributes only its ordered `{id,name}` projection after
the shared evidence message. It is never occurrence evidence.
for an operation and contributes names-only grounding after the shared evidence
message. Notarius resolves the model's selected name into the unchanged exact
durable ID/name pair. It is never occurrence evidence.
Each occurrence must use one exact registry ID/name pair. An extraction response
with an unknown ID or mismatched name is rejected as invalid model output; the
configured pipeline may retry it and never accepts a partial artifact.
with an unknown or ambiguous selected name is rejected as invalid model output;
the configured pipeline may retry it and never accepts a partial artifact.
Normalization and validation remain defense in depth for artifacts entering
through other boundaries: normalization canonicalizes a recognized name by ID,
preserves unknown values for the registry validator, and the registry validator

View File

@@ -85,9 +85,10 @@ for later artifacts.
`dnd/item-occurrences` requires one approved item registry through its
`item_registry` reference slot for both extraction and normalization. Its
consumer receives only an ordered, source-free `{id,name}` projection; the
registrys source references are never occurrence evidence. Unknown IDs and
mismatched pairs are rejected by the occurrence contract. See the
consumer receives names-only grounding; Notarius resolves the selected name
into the unchanged exact durable ID/name pair. The registrys source references
are never occurrence evidence. Unknown or ambiguous selections are rejected by
the occurrence contract. See the
[item-occurrence artifact](dnd-item-occurrence-artifacts.md) for that strict
wire contract, [Configuration](../config.md#d-d-reference-slots) for binding
rules and validator selection, and the [JSON output contract](json-output.md)

View File

@@ -73,10 +73,12 @@ complete canonical evidence sequence.
Both extraction and normalization require exactly one `location_registry` reference of
kind `dnd/location-registry`, media type `application/json`, and at most 1 MiB. The
registry provides identity grounding only: unknown IDs and mismatched ID/name
pairs are rejected rather than guessed or reassigned. The current transcript is
the only evidence source for an occurrence; registry evidence and provenance
never become occurrence evidence.
registry provides identity grounding only. The model selects a supplied
contextual name-and-registry-reference descriptor, and Notarius resolves it
into the exact durable ID/name pair. Unknown, partial, or ambiguous selections
are rejected rather than guessed or reassigned. The current transcript is the
only evidence source for an occurrence; registry evidence and provenance never
become occurrence evidence.
See [Configuration](../config.md#d-d-reference-slots) for the selectable slot
and generated-handoff compatibility, [D&D module internals](../internal/dnd.md)

View File

@@ -83,9 +83,11 @@ not evidence for later artifacts.
## Consumers and publication
`dnd/location-occurrences` requires one approved location registry through its
`location_registry` reference slot. Its prompt receives an ordered source-free `{id,
name}` projection and must not treat registry references as occurrence
evidence. See the [location-occurrence artifact](dnd-location-occurrence-artifacts.md)
`location_registry` reference slot. Its prompt receives contextual selectors
containing a canonical name and registry references; Notarius resolves a
selection into the unchanged exact durable ID/name pair. Registry references
must not be treated as occurrence evidence. See the
[location-occurrence artifact](dnd-location-occurrence-artifacts.md)
for that contract, [Configuration](../config.md#references-and-ordered-handoffs)
for binding rules, and the [JSON output contract](json-output.md) for
publication.

View File

@@ -67,10 +67,12 @@ for uncertain classification.
## Identity, evidence, and order
The required normalized [NPC registry artifact](dnd-npc-registry-artifacts.md) resolves
the exact `{npc_id, name}` pair. Unknown IDs and names that do not match their
ID are rejected; normalization does not repair names by similarity. Registry
references are provenance only and never replace an occurrence's own evidence.
The required normalized [NPC registry artifact](dnd-npc-registry-artifacts.md)
supplies names-only contextual grounding to the model. Notarius resolves the
selected name and writes the exact `{npc_id, name}` pair. An unknown or
ambiguous selection rejects the complete model result; normalization does not
repair names by similarity. Registry references are provenance only and never
replace an occurrence's own evidence.
The registry may include an identity established by a factual third-party
mention; that provenance alone does not create a `mentioned` occurrence. Each
occurrence remains a separately cited fact in the current transcript.

View File

@@ -82,9 +82,10 @@ with its own cited evidence and category.
This registry can ground actor or caster names in the [spell](dnd-spell-artifacts.md)
and [combat-turn](dnd-combat-turn-artifacts.md) artifacts. It is required to
resolve the canonical `name` in an [NPC occurrence](dnd-npc-occurrence-artifacts.md).
Occurrence consumers receive an ordered source-free `{id,name}` projection;
spells, combat turns, and the [enemy-event artifact](dnd-enemy-event-artifacts.md)
receive names-only grounding for actor or subject display. None of these
Occurrence consumers receive names-only grounding; Notarius resolves the
selected canonical name and writes the unchanged exact durable ID/name pair.
Spells, combat turns, and the [enemy-event artifact](dnd-enemy-event-artifacts.md)
also receive names-only grounding for actor or subject display. None of these
projections supply later-artifact evidence. [Configuration](../config.md#d-d-reference-slots)
owns the `npc_registry` binding rules.
The [JSON output contract](json-output.md) defines publication, and

View File

@@ -67,6 +67,12 @@ including a collision with the embedded catalog. Matching uses the catalogs
case, whitespace, and apostrophe normalization, so authors should avoid names
or aliases that normalize to another spell.
Spell extraction receives the effective catalog as deterministic canonical-name
and alias pairs. An alias in the transcript selects its associated canonical
name; the extractor is instructed to return that canonical spelling. The
projection contains no catalog source metadata or provenance, and aliases
remain recognition context rather than transcript evidence.
The overlay is a recognition aid only. The durable spell-artifact schema and
source-evidence rules are defined by the
[D&D spell artifact contract](dnd-spell-artifacts.md).

View File

@@ -1,9 +1,11 @@
# Published Evidence Context
This contract defines the optional `source/evidence-context` artifact emitted
by the production JSON output. Its configuration is owned by
[Configuration](../config.md#module-bindings-and-validators); its logical-file
discovery is owned by [Published JSON Output](json-output.md).
by the production JSON output. It is a selected source-unit excerpt for
convenient reading alongside normalized lane artifacts; it is not a second
citation or provenance model. Its configuration is owned by
[Configuration](../config.md#module-bindings-and-validators), and its
logical-file discovery is owned by [Published JSON Output](json-output.md).
## Identity And Discovery
@@ -26,91 +28,80 @@ its absence means evidence publication was not enabled for that bundle.
## Payload
The v1 payload is a JSON object with required `source_id`, `source_digest`,
`window_units`, `selected_lanes`, and `contexts` fields. `selected_lanes` and
`contexts` are always arrays; an enabled configuration with no accepted direct
evidence publishes `contexts: []`.
The v1 payload is a top-level JSON array of generic source units. There is no
wrapper, source-level metadata, context grouping, lane identifier, or evidence
reference in the payload. An enabled configuration with no contributing
accepted evidence publishes `[]`.
```json
{
"source_id": "session-alpha",
"source_digest": "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef",
"window_units": 1,
"selected_lanes": ["npc_registry", "spells"],
"contexts": [
{
"context_ref": {
"source_id": "session-alpha",
"start_unit_id": 10,
"end_unit_id": 20
},
"evidence_refs": [
{
"lane_id": "spells",
"source_ref": {
"source_id": "session-alpha",
"start_unit_id": 10,
"end_unit_id": 10
}
}
],
"units": [
{
"id": 10,
"kind": "transcript_segment",
"text": "Aria casts Cure Wounds.",
"ref": {
"source_id": "session-alpha",
"start_unit_id": 10,
"end_unit_id": 10
}
},
{
"id": 20,
"kind": "transcript_segment",
"text": "The party regroups.",
"ref": {
"source_id": "session-alpha",
"start_unit_id": 20,
"end_unit_id": 20
}
}
]
[
{
"id": 10,
"kind": "transcript_segment",
"text": "Aria casts Cure Wounds.",
"ref": {
"source_id": "session-alpha",
"start_unit_id": 10,
"end_unit_id": 10
}
]
}
},
{
"id": 20,
"kind": "transcript_segment",
"text": "The party regroups.",
"ref": {
"source_id": "session-alpha",
"start_unit_id": 20,
"end_unit_id": 20
}
}
]
```
Each context requires a `context_ref` object and `evidence_refs` and `units`
arrays. `context_ref` identifies the first and last included unit. Each
evidence entry contains a selected `lane_id` and an original `source_ref`. A
unit uses the existing source-unit shape: required `id`, `kind`, `text`, and
self `ref`, plus optional JSON-object `metadata`. Fixed payload objects reject
unknown fields; unit metadata may contain application-defined JSON values.
Each source unit has required `id`, `kind`, `text`, and self `ref` fields.
`ref` contains `source_id`, `start_unit_id`, and `end_unit_id`, and both unit
endpoints identify that unit's `id`. A unit may also contain source-owned
`metadata`, an open-ended JSON object. Fixed unit and reference fields are
strict: consumers must reject unknown fixed fields, malformed units, invalid
self-references, units whose `source_id` differs from other units in the same
excerpt, and a payload that is not the array described here.
## Citations And Context
The excerpt preserves each selected unit exactly as represented by the
validated generic source document. It does not add evidence-context-specific
annotations or reshape source-owned metadata.
`evidence_refs` are the authoritative citations. They identify the direct
references emitted by accepted normalized artifacts. `context_ref` and the
units collection include those cited units plus nearby source units selected by
the configured window. They are explanatory context, not widened citations.
## Selection And Citations
Only accepted outputs from the configured lane allowlist contribute. Rejected,
failed, absent, and lane-filtered outputs do not contribute. The artifact never
contains raw input bytes, prompts, model responses, auxiliary reference
content, credentials, or filesystem paths.
The framework obtains direct source references only through typed evidence
projections of accepted normalized artifacts in the configured lane allowlist.
It validates each reference against the current source document, expands its
range by `window_units` source-unit positions on each side, clamps at document
boundaries, and takes the union of all expanded ranges. The output contains
each selected source unit once in source-document position order, regardless
of numeric unit IDs. Repeated references, overlapping windows, and citations
from multiple lanes do not duplicate a unit. Rejected, failed, absent,
inactive, and unselected lanes contribute nothing.
## Ordering And Compatibility
Normalized lane artifacts remain authoritative for citations and for which lane
cited a range. The excerpt has no lane attribution and must not be used to
reconstruct it. Its included nearby units provide reading context only; they
do not widen any citation in a lane artifact.
The selected lane allowlist is lexical. Contexts and units are in source
document position order, not numeric unit-ID order. Direct evidence entries
are deterministically ordered by lane and source reference. Overlapping or
contiguous windows merge, and each source unit appears at most once in the
resulting contexts.
The excerpt contains at most every generic source unit once. It can therefore
equal the complete generic source document when coverage is broad or the
window is large. No byte-, token-, or compression-size guarantee is made, and
the framework does not truncate the excerpt to meet an arbitrary size limit.
## Consumer Responsibilities And Data Handling
The artifact is additive to the JSON bundle and is not a lane payload,
normalized-output count, checkpoint, or generated reference. Consumers that
do not need it must tolerate the absent optional descriptor. Consumers that do
use it should preserve the artifact and its schema identity with the run
provenance, and should treat its source text and metadata as sensitive durable
content.
do not need it must tolerate an absent descriptor. Consumers that do use it
should validate the descriptor and payload before use, retain the artifact with
its schema identity when needed for a run record, and read citations from the
corresponding normalized lane artifacts.
The excerpt contains source-unit text and source-owned metadata and is durable
output. Treat it as sensitive source content, apply appropriate access controls
and retention, and do not assume its selected form is materially smaller or
less sensitive than the original input.

View File

@@ -24,7 +24,7 @@ root for the logical discovery described here.
| `warnings.json` | Accepted-output and run warnings. |
| `lanes/<safe-lane-id>.json` | One normalized artifact payload for each lane. |
| `chunk-map.json` | Optional accepted chunk map, when its export is enabled and available. |
| `evidence-context.json` | Optional source-context artifact, when evidence publication is enabled. |
| `evidence-context.json` | Optional selected source-unit excerpt, when evidence publication is enabled. |
JSON files are pretty-printed with a trailing newline. Lane payloads are
accepted only when their media type is `application/json`.

View File

@@ -1,11 +1,11 @@
# PromptKit Integration
Notarius pins
[`gitea.maximumdirect.net/eric/promptkit` v0.5.0](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0)
[`gitea.maximumdirect.net/eric/promptkit` v0.8.0](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.8.0)
as its in-process prompt engine. The upstream
[Go package consumer guide](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0/docs/consumers/pkg-promptkit.md)
[Go package consumer guide](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.8.0/docs/consumers/pkg-promptkit.md)
owns the public engine API, and the upstream
[format reference](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0/docs/formats.md)
[format reference](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.8.0/docs/formats.md)
owns prompt, profile, and schema file contracts.
## Supported Boundary
@@ -26,7 +26,7 @@ Notarius relies on the root `promptkit` package to:
admission exhaustion through `ErrCapacityExceeded`.
The pinned
[`BackendLocal`, `LocalBackend`, and `WithBackend` API](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0/backends.go)
[`BackendLocal`, `LocalBackend`, and `WithBackend` API](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.8.0/backends.go)
owns the registration and backend-capacity contract.
For one completion, the adapter calls `PrepareExecution`, takes a
@@ -52,7 +52,7 @@ Notarius sends one stable effective session through PromptKit's direct session
field, which is authoritative for provider session behavior. It also retains
the same value as the `session_id` prompt variable for maintained prompt
compatibility. The generated identifier is 76 ASCII characters, within
PromptKit v0.5.0's 256-code-point session limit. Session IDs are non-secret
PromptKit v0.8.0's 256-code-point session limit. Session IDs are non-secret
correlation identifiers and may be exposed to providers and provider
observability. The CLI contract owns generation and override behavior.
@@ -84,7 +84,13 @@ configuration and deployment workflow are defined in
[Configuration](../config.md#promptkit-profiles) and
[Operations](../operations.md#promptkit-profile-deployment).
Notarius supports this boundary against PromptKit v0.5.0. Its fallback source,
PromptKit owns `base_profile` resolution under its
[pinned format rules](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.8.0/docs/formats.md).
Notarius records the selected leaf identity and resolved target without parsing
or merging inheritance. An unset filesystem `api_key_env` is optional and may
reach the provider without authorization, which can result in a 401 or 403.
Notarius supports this boundary against PromptKit v0.8.0. Its fallback source,
prepared-execution, inspection, and typed capacity APIs are used as public
upstream contracts; other PromptKit APIs or file-format behavior are not
implicitly supported. A dependency upgrade requires reviewing the adapter,
@@ -97,6 +103,10 @@ pinned upstream documentation.
module assets, maps its transport-neutral completion contract, prepares and
executes requests, validates output, records provenance, captures debug
material, redacts errors, and preserves timeout ownership.
PromptKit provider error details do not cross the ordinary completion boundary.
Notarius exposes a provider-neutral generation category and optional status;
redacted provider details are retained only in requested debug material.
[D&D Module Internals](../internal/dnd.md) owns the embedded
`dnd-extraction` fallback profile and the maintained D&D prompt defaults.
[Configuration](../config.md#promptkit-profiles) defines how a Notarius
@@ -105,4 +115,7 @@ the conventional local backend.
PromptKit API or format changes outside this boundary are not implicitly
supported. Updating the pinned version requires reviewing the adapter and
profile/configuration contracts against the upstream documentation.
profile/configuration contracts against the upstream documentation. Maintained
production prompts use PromptKit's bounded structural-repair contract; their
current declaration is one additional repair attempt. Notarius retains the
transport-neutral boundary and does not expose PromptKit types to modules.

View File

@@ -24,10 +24,14 @@ preparation, and runner mechanics after their inputs are supplied.
## Dispatch And Configuration Handoff
The root dispatcher handles help, configuration validation, pipeline listing,
and a pipeline run. It normalizes injectable options before dispatch so that a
missing production dependency fails as a command error rather than reaching
execution.
The root dispatcher handles help, version reporting, configuration validation,
pipeline listing, and a pipeline run. Version reporting resolves build
information through `internal/buildinfo` before production composition, so the
diagnostic remains available without configuration or runtime collaborators.
The public syntax, streams, exit classes, and version semantics are defined by
the [CLI reference](../cli.md). Other root commands normalize injectable
options before dispatch so that a missing production dependency fails as a
command error rather than reaching execution.
Commands that need configuration use one shared loader. The CLI discovers the
file, parses it through **internal/core/config**, starts from defaults, applies

View File

@@ -84,13 +84,15 @@ replace it with a complete profile of the same ID from the configured PromptKit
source. Deployment profile selection is documented in
[Configuration](../config.md#promptkit-profiles).
The transcript assets have distinct consumers. Scene chunking consumes the
complete-session `common-dnd-transcript-full.md`; extraction prompts consume
the current-chunk `common-dnd-transcript-chunk.md`; and NPC, location, and item
normalization consume `common-dnd-transcript-windows.md` alongside their
candidate collections. Player, party, glossary, and compatible campaign
references provide disambiguating context, not evidence. Reference material is
canonically ordered before rendering so equivalent inputs remain stable.
The D&D transcript assets have distinct consumers. Scene chunking consumes the
complete-session `common-dnd-transcript-full.md`, while extraction prompts
consume the current-chunk `common-dnd-transcript-chunk.md`. NPC, location, and
item normalization instead mount the generic semantic-reconciliation
candidate and transcript-window presentation assets. Player, party, glossary,
and compatible campaign references provide disambiguating context only when
declared by the active prompt; they never establish evidence. Reference
material is canonically ordered before rendering so equivalent inputs remain
stable.
Extraction prompts render the common system and identity messages first, then
cached campaign references and the cached chunk transcript. Evidence policy and
@@ -100,10 +102,11 @@ reusable extraction prefix identical while preserving the lane-specific suffix.
Scene chunking intentionally uses a different order: system, cached campaign
references, uncached module instructions, then the final ephemeral full
transcript. Entity normalization also has its own order: system, uncached
module instructions, ephemeral reconciliation policy, uncached candidates, and
final ephemeral transcript windows. These orders and cache controls are prompt
behavior; change them only through the owning manifest and prompt declaration.
transcript. Entity normalization also has its own order: D&D system, mandatory
generic protocol, ephemeral domain semantic instructions, generic candidate
presentation, and final ephemeral generic transcript windows. These orders and
cache controls are prompt behavior; change them only through the owning
manifest and prompt declaration.
## Evidence, Candidates, And Normalization
@@ -116,7 +119,8 @@ result.
Default chains keep responsibilities separate: structural validators assess the
candidate, source-reference validators resolve cited ranges against the current
source, durable-schema validation checks an approved representation, and
source and require extraction evidence to stay within the current chunk,
durable-schema validation checks an approved representation, and
relatedness validators report advisory evidence concerns. The configured order
is documented in
[Configuration](../config.md#production-validator-keys-and-default-chains).
@@ -127,14 +131,51 @@ combine results from distinct scenes, so it intentionally does not apply that
rule. Configuration owns the exact validator key and chain position.
Normalizers are deterministic for spells, combat turns, item occurrences, NPC
occurrences, scene descriptions, enemy events, and location occurrences. They canonicalize display
values and evidence, use source-document order for stable output, and issue
bounded warnings for changes or collapsed duplicates. The NPC and location
normalizers are intentional exceptions: each first produces a deterministic
candidate set, then may use a bounded structured-LLM proposal to reconcile
identity groups. Invalid or unusable proposals retain the deterministic result
and surface retry or fallback diagnostics; the model does not directly replace
durable records.
occurrences, scene descriptions, enemy events, and location occurrences. They
canonicalize display values and evidence, use source-document order for stable
output, and issue bounded warnings for changes or collapsed duplicates. NPC,
item, and location registry normalizers are intentional exceptions: each first
produces a deterministic candidate set, then may use a bounded structured-LLM
proposal to reconcile identity groups.
## Semantic Registry Reconciliation
The three registry normalizers instantiate the domain-neutral
`internal/framework/semanticreconcile` engine with default bounds. Each
eligible candidate receives a contiguous, one-based `candidate_id` for that
request. The model sees that handle, the candidate label and source-free
evidence ranges, plus bounded transcript windows; it returns only duplicate
groups of supplied handles and one supplied canonical handle per group. It
never returns names, evidence, durable IDs, or replacement records. Identical
labels and evidence remain independently selectable because their handles are
distinct.
The generic core owns the mandatory handle protocol, candidate and transcript
presentation, the private response schema, source-reference validation,
candidate and combined-material limits, structured completion, proposal
assessment, stable group ordering, and typed plan-application mechanics. The
D&D prompt contributes its system message and registry-specific semantic
instructions. The generic registrar registers the shared prompt and schema;
the D&D registrar registers each consuming prompt and the fallback profile.
Fewer than two eligible candidates skips the LLM without a semantic warning.
An exceeded bound also skips the call and preserves the deterministic
preprocessed registry, adding the registry's bounded fallback warning. Invalid
structured output or discarded proposal groups use the normalizer's existing
retry contract; retry exhaustion preserves the safe deterministic or
partially applied result and emits its bounded fallback warning. Provider,
transport, cancellation, and context-material failures remain execution
errors.
Application remains typed and registry-owned. All three policies select the
canonical member's normalized display name, union member evidence in source
order, preserve ungrouped records, and derive durable identity only after
consolidation. NPC IDs derive from the final name. Item IDs also derive from
the final name, and a typed guard prevents currency aliases from crossing
denominations or mixing currency with non-currency records. Location IDs
derive from the final name and final evidence, preserving same-name,
parent/child, and distinct physical-place identities. Registry warning scopes,
reason codes, and postconditions remain outside the generic core.
## Generated References And Grounding
@@ -144,11 +185,19 @@ producer provenance; consumers resolve the handed-off artifact into an
immutable, validated projection for each operation. External files are checked
during preparation, while generated artifacts are resolved at the handoff.
NPC, location, and item registries project ordered, source-free `{id, name}`
pairs to their respective occurrence extractors and normalizers. Exact ID/name
matching preserves every identity the registry recognizes, including same-name
locations with distinct source anchors. The NPC registry additionally supplies
names-only actor grounding to spells, combat turns, and enemy events.
NPC and item registry consumers receive names-only grounding. Location
consumers receive a contextual selector containing the canonical name and the
registry references needed to distinguish same-name places. The calling module
resolves those supplied selections locally and maps them into the unchanged
durable ID/name pair; an unknown or ambiguous selection rejects the complete
occurrence result rather than accepting a partial mapping. The NPC registry
additionally supplies names-only actor grounding to spells, combat turns, and
enemy events.
Registry references establish a registry identity and may disambiguate a
selection, but never become occurrence evidence. Each occurrence keeps its own
current-transcript source references, even when it was grounded through the
same registry record.
Scene descriptions are eligibility-only projections: they retain current-chunk
classification data, not scene prose or evidence, and exist to route combat
extraction. Enemy-event extraction also projects combat turns to `actor` and
@@ -170,7 +219,7 @@ checkpoint fingerprint.
| Spells | May use a spell-catalog overlay and optional NPC grounding; the catalog validator supplies domain-specific semantic checks. |
| NPC registry | Establishes transcript-grounded NPC identities, including factual third-party mentions, without assigning occurrence categories. It does not consume an NPC registry, and its normalizer is the LLM-assisted reconciliation exception described above. |
| Combat turns | Requires a scene-description artifact. It calls the LLM only for an exact `combat` classification; exact non-combat classifications return an accepted empty result, while missing or mismatched classifications return an empty result with a bounded warning. Optional NPC grounding never becomes evidence. |
| Item occurrences | Requires the normalized item registry for exact ID/name grounding at extraction and normalization. Campaign context may disambiguate, but the registry never becomes occurrence evidence. |
| Item occurrences | Requires the normalized item registry for exact deterministic grounding at extraction and normalization. Campaign context may disambiguate, but the registry never becomes occurrence evidence. |
| Item registry | Produces source-grounded item types and unique designations. Its LLM-assisted reconciliation is proposal-only, preserves distinct currency denominations and item types, and does not create per-instance identities. |
| NPC occurrences | Requires the normalized NPC registry at extraction and normalization, using it for canonical actor grounding only. It separately emits cited current-transcript occurrence facts, including `mentioned`, rather than deriving them from registry provenance. |
| Scene descriptions | Produces the classifications consumed by combat routing; it does not consume an NPC registry or provide evidence for combat artifacts. |

View File

@@ -24,6 +24,9 @@ adapter does not own source evidence, artifact conversion, normalization, or
durable schemas. Those responsibilities remain with the module and its
[integration contract](../integrations/).
The calling module also resolves contextual entity selections and attaches any
application identity; PromptKit and this adapter do not own entity identity.
`PromptKitClient` validates the request target and prompt identity, maps each
named material to a PromptKit inline artifact while preserving its origin URI,
passes the supplied request session through to PromptKit's direct per-run
@@ -72,7 +75,7 @@ backend membership as runtime without performing generation. Fallback assets
are mounted only when at least one source is registered. The production D&D
registrar contributes its `dnd-extraction` fallback, and the maintained D&D
prompts select that logical ID by default. PromptKit owns source precedence and
profile parsing: an operator-provided matching profile takes precedence over a
profile parsing and inheritance: an operator-provided matching profile takes precedence over a
fallback profile without Notarius merging either document.
When the registration is absent, a profile selecting `backend: local` fails
inspection instead of falling back to a built-in or endpoint-only target.
@@ -99,7 +102,8 @@ because it changes scheduling rather than execution semantics.
Production construction creates one PromptKit client and wraps it in one
scheduled client. The scheduler has a fixed, positive permit limit, serves
queued calls in FIFO order, and removes a queued call when its context is
cancelled. A granted permit is released exactly once on every completion path.
cancelled. It rechecks the caller context after admission and before dispatch.
A granted permit is released exactly once on every completion path.
The scheduled wrapper surrounds every `CompleteStructured` call, so concurrent
lanes, pipeline retries, and LLM-backed validators share the same provider-call
@@ -138,12 +142,28 @@ arrangement and its data-only boundary are defined by
[ADR-0011](../adr/0011-centralize-llm-assets.md), rather than by this runtime
guide.
The generic registrar is the sole production registration owner for the
semantic-reconciliation default prompt and private response schema. The
domain-neutral reconciliation package also exposes only its mandatory protocol
and candidate/transcript presentation files for domain prompt manifests. D&D
registry normalizers mount those files while retaining ownership and hashing
of their D&D system message, semantic instructions, and complete prompt
declaration. The response schema is therefore registered once even though
several typed normalizers select it.
Mounted prompt assets determine a module's fingerprint. The fingerprint hashes
only the module and shared files explicitly selected by its manifest, so an
unrelated asset does not invalidate a checkpoint. Schema loaders validate JSON,
attach identity and digest metadata, make defensive copies, and expose
diagnostics without raw schema bytes.
Semantic-reconciliation normalizers extend this identity with the shared
response-schema digest, framework policy version, and complete limit-policy
digest. Their manifest metadata records the same content-free prompt, schema,
policy, and limit identities together with domain identity and normalization
policies. Request-local handles, source material, proposal content, and raw
asset bytes are not checkpoint metadata.
Private response schemas validate a model transport envelope. They are not the
durable artifact schema and should not be documented as an external wire
contract. Durable formats and compatibility rules remain in the
@@ -176,7 +196,9 @@ structured-output validation. The adapter reports an empty result, validation
failure, empty structured body, or decode failure as
`ErrInvalidStructuredOutput`, while retaining the returned raw bytes and debug
material when they exist. Provider failures remain operational errors rather
than output-validation failures.
than output-validation failures. Apart from documented context, capacity, and
invalid-output categories, provider error values and types do not cross the
adapter error chain; callers receive only a credential-redacted diagnostic.
When PromptKit rejects backend admission before generation, the adapter maps
`promptkit.ErrCapacityExceeded` to
@@ -188,14 +210,29 @@ caller context takes precedence. The adapter does not retry capacity failures;
the pipeline's existing binding attempt policy sees the operational error and
decides whether to rerun the complete operation.
Prompt-declared repair is executed within PromptKits structured-output flow.
The current production D&D prompt manifests set repair attempts to zero. That
setting does not replace pipeline retry behavior: a bindings configured retry
count reruns its stage attempt after an error or rejection, and an exhausted
rejection is a recorded output rather than a provider error. The pipeline owns
attempt lifecycle, validation chains, and retry diagnostics; see
[Pipeline Internals](pipeline.md#validation-retries-and-output) and the
[binding reference](../config.md#module-bindings-and-validators).
PromptKit executes structural repair within its structured-output flow. The
maintained production prompt manifests declare one additional repair attempt.
When a resolved binding supplies a repair value, the adapter inspects the
prompt, copies its complete output contract, changes only the repair limit, and
passes that complete replacement contract to PromptKit. This preserves the
prompt's output format, validation mode, schema, and provider structured-output
settings.
A successful repair is an ordinary successful completion, not a warning. The
adapter reports PromptKit's actual repair count and its cumulative usage
directly, without adding the initial and corrective counts again. Debug prompt
material records the configured complete contract; debug response material
records the repaired response and actual validation result. If the repair
budget is exhausted, the adapter retains the final raw bytes and debug material
and reports `ErrInvalidStructuredOutput`. Generation failures during an initial
or corrective call remain provider-neutral operational errors with the same
redaction boundary.
Structural repair does not replace pipeline retry behavior: a binding's
configured retry count reruns its complete stage attempt after an error or
rejection. The pipeline owns attempt lifecycle, validation chains, and retry
diagnostics; see [Pipeline Internals](pipeline.md#validation-retries-and-output)
and the [binding reference](../config.md#module-bindings-and-validators).
## Timeout Ownership
@@ -234,6 +271,9 @@ redacted before it crosses the runtime boundary. Known-secret redaction is
available to other runtime collaborators; it does not make prompt or response
contents safe for general logging.
Generation failures expose an application-owned category and optional HTTP
status. Provider code, type, and message remain debug-only, after redaction.
## Failure Boundaries
- Construction fails for missing asset registries, mutually exclusive profile

View File

@@ -42,14 +42,23 @@ generic source references and must use the codec's exact Go type. It does not
interpret surrounding context or publish files; the pipeline validates the
capability during preparation and the output boundary owns publication. See
the [Published Evidence Context contract](../integrations/evidence-context.md)
for the durable result.
for the durable source-unit excerpt. Lane artifacts retain citation and lane
provenance; the framework does not add either to that published excerpt.
An artifact family is broader than a module: it owns the cohesive domain
feature across its artifact type, codec, stage modules, validators, prompt
policy, schemas, identity helpers, and reference projections. An extractor and
normalizer in one artifact family remain independently registered modules in
their respective pipeline stages. This ownership vocabulary does not create a
new registry or change the fixed pipeline.
## Production Composition
Production composition is intentionally split by family:
- The generic registrar provides the unit chunker, generic JSON validators,
and JSON output encoder.
JSON output encoder, and shared semantic-reconciliation prompt and response
schema assets.
- The Seriatim registrar provides the transcript input adapter. Its external
input behavior is defined by the [Seriatim contract](../integrations/seriatim.md).
- The D&D registrar provides its codecs, extractors, mergers, normalizers,
@@ -60,6 +69,36 @@ The CLI owns the composition that invokes these registrars. A module package
may register its own family but must not assemble the CLI or make framework
packages depend on production extensions.
## Semantic Reconciliation
`internal/framework/semanticreconcile` is a domain-neutral strategy used by a
typed normalize module; it is not itself a selectable stage module. A
source-backed artifact-family normalizer projects its deterministic records
into contextual candidates and owned typed record envelopes, supplies its
chosen prompt identity and resolved LLM profile, and constructs an engine with
explicit limits. The core filters invalid evidence, assigns contiguous
request-local integer handles, renders bounded candidate and transcript
materials, invokes the structured-completion boundary, and assesses the
returned duplicate groups into a stable non-overlapping plan.
The normalizer then applies that plan through a typed `ApplicationPolicy`. The
core preserves ungrouped records, contribution order, and provenance while the
artifact family owns group guards, field and evidence consolidation, durable
ID derivation, retry and fallback presentation, warnings, and postconditions.
Request-local handles do not enter the typed value or durable artifact. Fewer
than two eligible candidates skips model invocation; exceeding a candidate or
combined-material bound preserves the deterministic result under the family's
fallback policy. Provider, transport, cancellation, and context-construction
failures remain execution errors.
The core supplies a conservative generic prompt and the single private
response schema. A domain prompt may substitute its semantic instructions but
mounts the core-owned protocol and candidate/transcript presentation assets.
Prompt, schema, policy, and limit identities participate in manifest metadata
and checkpoint fingerprints. The generic registrar owns production
registration of those shared assets; a consuming domain registrar owns only
its domain prompt.
## Adding Or Changing A Module
1. Choose the pipeline stage and the typed artifact boundary. Put external

View File

@@ -25,10 +25,12 @@ physical state roots.
| Area | Implemented owners | Responsibility |
| --- | --- | --- |
| Executable and command boundary | **cmd/notarius**, **internal/cli** | Process entry, command dispatch, configuration discovery, production composition, runtime collaborator setup, durable file placement, and user-facing reporting. |
| Build information | **internal/buildinfo** | Resolves a stable linked release tag or build metadata for the diagnostic root version command. |
| Configuration | **internal/core/config** | Defaults, strict YAML parsing, environment overrides, structural validation, effective resolution, redaction, and resolved-composition summaries. |
| Generic models | **internal/core/source**, **internal/core/artifacts**, **internal/framework/contracts** | Source documents and chunks, manifests and provenance, plus typed artifact, reference, validation, output, and structured-completion contracts. |
| Pipeline framework | **internal/framework/pipeline** | Registries, profile and reference resolution, typed preparation, validation, retry coordination, ordered execution, handoff, and result assembly. |
| LLM and prompt runtime | **internal/framework/llm**, **internal/framework/promptfs** | Provider-neutral structured completions, scheduling, profile recording, prompt assets, schema registration, and credential-shaped-value redaction. |
| Semantic reconciliation | **internal/framework/semanticreconcile** | Bounded source-backed candidate preparation, request-local handle proposals, deterministic assessment, typed plan application, and reconciliation identity metadata; see [Module Internals](modules.md#semantic-reconciliation) and [D&D Module Internals](dnd.md#semantic-registry-reconciliation). |
| Embedded LLM content | **assets** | Read-only centralized LLM-facing content, scoped by its consuming package; see [LLM Runtime](llm.md#prompt-and-schema-assets) and [D&D Module Internals](dnd.md#prompt-construction). |
| Runtime state | **internal/core/fileio**, **internal/core/debugbundle**, **internal/framework/checkpoint**, **internal/framework/chunkplan**, **internal/framework/chunkmap**, **internal/framework/debug** | Confined atomic files, debug bundles, checkpoint and chunk-plan state, accepted chunk maps, and pipeline-facing debug recording. |
| Production extensions | **internal/modules/generic**, **internal/modules/seriatim**, **internal/modules/dnd** | Domain-neutral extensions, Seriatim input support, and D&D extraction families registered into the production catalog. |
@@ -49,8 +51,9 @@ the CLI composition boundary.
composition, and path safety.
- [LLM Runtime](llm.md): structured completion, scheduling, prompt assets,
profiles, and secret handling.
- [Module Internals](modules.md): generic extension registration, module
construction, validation, and reference mechanics.
- [Module Internals](modules.md): generic extension registration, artifact
families, module construction, semantic reconciliation, validation, and
reference mechanics.
- [D&D Module Internals](dnd.md): shared D&D extractor conventions, generated
reference projections, and lane-specific exceptions. Durable D&D and
Seriatim data shapes remain in the [integration contracts](../integrations/).

View File

@@ -36,9 +36,16 @@ assigns a deterministic resolved-composition digest. The resolved pipeline
contains bindings and declared reference targets, not external reference bytes.
After selection, the resolver applies command, binding, and pipeline profile
precedence to LLM-backed bindings and validators only; prompt defaults remain
an empty resolved binding profile. Deterministic bindings remain profile-free.
These effective values are part of the digest, so execution and checkpoint
consumers do not repeat profile inheritance.
an empty resolved binding profile. It resolves structural output repair
separately: a binding's `structured_output_repair_attempts` value wins, then a
pipeline value applies to LLM-backed bindings and validators, and omission
leaves the prompt-owned policy intact. An explicit repair value on a
deterministic binding is rejected. Resolved bindings own copied repair values,
and these effective values are part of the digest, so execution and checkpoint
consumers do not repeat profile inheritance or configuration resolution.
Each LLM request receives its own copy of that resolved value. PromptKit spends
it only for structural correction inside one completion; the runner's binding
retry policy remains the separate outer budget for complete stage attempts.
Configuration resolution supplies the selected profile and catalog; see
[Configuration Internals](configuration.md).
@@ -46,16 +53,18 @@ External reference materialization happens before preparation. The materializer
checks that each slot is declared by the selected module, resolves a file path
relative to the correct configuration or working-directory origin, reads
UTF-8 text, verifies media type and size limits, and retains bounded
provenance. A generated-artifact selector remains declared but has no bytes
until its producing step completes.
provenance. For a positive slot limit, it reads at most the limit plus one byte
and rejects overflow before retaining content. A generated-artifact selector
remains declared but has no bytes until its producing step completes.
Preparation is the construction boundary. It validates the resolved shape and
registry set, clones the resolved data, then constructs the input adapter,
chunker, stage-local validators, every typed lane, and output encoder with
cloned options, references, and shared dependencies. It also collects stable
checkpoint fingerprints. Missing registrations, incompatible typed entries,
nil implementations, and constructor failures are reported before source
parsing or any stage operation begins.
chunker, stage-local validators, every typed lane, and output encoder. Each
registered builder receives its own cloned build request immediately before its
module-owned code runs. Preparation also collects stable checkpoint
fingerprints. Missing registrations, incompatible typed entries, nil
implementations, and constructor failures are reported before source parsing
or any stage operation begins.
An output encoder can opt into source-evidence publication through its output
policy. Preparation keeps the configured lane allowlist and active lanes
@@ -115,9 +124,12 @@ for started workers, and prevents output encoding.
Every chunk, extract, merge, and normalize candidate passes its resolved
validator chain. Validators receive immutable canonical input appropriate to
their target: chunks, typed values, or serialized codec bytes. They may
approve, approve with warnings, reject, or fail. A rejection is an ordinary
pipeline result; a validator error is a framework error.
their target: chunks, codec-decoded typed candidates, or serialized codec
bytes. Each typed validator receives a newly decoded value from the one
candidate serialization for that attempt, while serialized validators receive
separately owned representation bytes and schema metadata. They may approve,
approve with warnings, reject, or fail. A rejection is an ordinary pipeline
result; a validator error is a framework error.
The runner applies the binding's retry policy around a stage operation and its
complete validation chain. It preserves warnings only from the final accepted

View File

@@ -47,6 +47,9 @@ The serialized
they do not describe a current public state surface.
Ordered-step lane checkpoints include the step identity in their storage scope.
Accepted step and lane identities are encoded injectively before becoming
filesystem path components, while ordinary safe identifiers retain their
readable paths.
When a later lane consumes a generated artifact, its dependency fingerprints
include the producer's artifact kind, complete schema identity, media type,
canonical content digest, and size. Ordinary resume compares those fingerprints

View File

@@ -5,6 +5,23 @@ This is the canonical guide for operating Notarius runtime state. The
[Configuration](config.md) owns fields, defaults, and precedence. Maintainers
who need implementation mechanics should read [Run State Internals](internal/state.md).
## Source Deployment
Linux is the supported deployment platform. Install a pinned source release
with the Go version declared in `go.mod` (currently Go 1.25.5):
~~~sh
GOWORK=off go install \
gitea.maximumdirect.net/eric/notarius/cmd/notarius@vMAJOR.MINOR.PATCH
~~~
Pin the exact tag in deployment automation rather than following a branch.
Use [`notarius --version`](cli.md#command-summary) as a diagnostic after
installation; its syntax and semantics are owned by the [CLI reference](cli.md).
The maintainer publication process, including tag guards and verification,
belongs to [Source Releases](release.md). macOS builds are best-effort for
development, and Windows is unsupported.
## State Surfaces
Each run can use independent roots with different retention and access-control
@@ -72,6 +89,9 @@ provider call or credentials:
notarius config validate --config /etc/notarius/config.yml --pipeline dnd-session
~~~
An unset optional `api_key_env` reaches the provider without authorization and
may receive a 401 or 403 response.
Profile paths are currently resolved from the process working directory, not
from the configuration file. The complete example's
`./examples/profiles/dnd-extraction.yml` path is valid for a repository-root
@@ -110,9 +130,9 @@ are defined in [Accepted Chunk Map](integrations/chunk-map.md). An optional
[evidence context](integrations/evidence-context.md) contains source-unit text
and metadata. It is not a cache or debug artifact: retain it with the output
bundle only for as long as consumers need it, and apply source-content access
controls to the entire bundle. Selected lanes may collectively cite most of a
transcript, so a broad allowlist can make the evidence artifact nearly as
sensitive and large as the source itself.
controls to the entire bundle. Its selected source-unit excerpt may include
every source unit once when coverage is broad or its configured window is
large, so do not assume a byte or token reduction or reduced sensitivity.
## Chunk-Plan Cache
@@ -255,14 +275,29 @@ Provider execution settings and the generation timeout come from the selected
PromptKit profile. The invocation-only **--reasoning-effort** and
**--clear-reasoning-effort** controls may replace or clear that profile setting
for all LLM-backed calls in one run without changing the profile. PromptKit
v0.5.0 does not add a provider retry loop. Notarius binding retries rerun the
complete module operation and validation chain as defined by
[module bindings](config.md#module-bindings-and-validators).
structural output repair happens within one structured-completion call. Its
effective `structured_output_repair_attempts` limit is resolved from the
selected binding, then the pipeline, then the prompt declaration; see
[module bindings](config.md#module-bindings-and-validators). This is distinct
from Notarius binding **retries**, which rerun the complete module operation
and validation chain and do not consume or replenish the structural-repair
limit. The maintained production prompts declare one repair attempt, paid only
after a structural failure. One structured completion with repair budget **R**
makes at most **R + 1** serial provider calls. If one stage attempt performs
**C** structured completions, a binding with **retries: N** has a maximum of
**(N + 1) * C * (R + 1)** provider calls; LLM-backed validators have their own
corresponding invocation counts and budgets. This is an upper bound, not a
promise that every call reaches a provider.
Timeouts are layered. Caller cancellation is the outer authority. A positive
effective generation timeout adds an inner request deadline, while zero
disables only that generation deadline. The HTTP client timeout remains a
transport-wide cap. Notarius does not add another timeout around PromptKit.
Repairs are serial within the same caller context, so their worst-case latency
and cost follow the provider-call bound above; provision run deadlines and
provider budgets accordingly. Credentials remain optional unless the selected
PromptKit profile requires one, in which case preparation fails before a
provider call when its configured credential is unavailable.
The pinned upstream boundary and profile-format links are in
[PromptKit Integration](integrations/pkg-promptkit.md).
@@ -279,6 +314,11 @@ positive value makes the effective active local-generation bound the smaller
of **total_llm** and that local limit, so a local limit of four permits no more
than four active local generations.
The Notarius scheduler admits one logical structured completion and holds that
permit while PromptKit performs its serial corrective calls. PromptKit applies
its selected-backend admission to each provider call; Notarius does not
reacquire a permit or add another scheduler for a repair.
For a positive local limit, PromptKit owns its default waiting capacity and
admission behavior. When a PromptKit backend has admitted all active and queued
work, a new call fails as capacity exhaustion before generation. The adapter

View File

@@ -24,6 +24,12 @@ DAGs or a general workflow language. Every stage remains explicit; general
chunking, merging, or normalization behavior must not be hidden inside an
extractor.
A stage module is one configured implementation of one pipeline stage. An
artifact family is the cohesive domain feature that owns an artifact across
the explicit stages and supporting codecs, validators, prompts, identity
rules, and reference projections. Artifact-family ownership does not combine
stages or alter the fixed pipeline.
Input and chunking are pipeline-wide. Each selected artifact lane owns its
extract, merge, and normalize stages, and the output stage aggregates the run's
lane outcomes.
@@ -39,6 +45,12 @@ implementations. Domain-neutral model and framework layers provide reusable
policy, contracts, and orchestration. Concrete input, pipeline, output, and
validation extensions depend inward on those generic layers.
Semantic reconciliation is one such domain-neutral framework mechanism. It
prepares bounded source context, invokes a shared model-judgment protocol,
validates proposals, and applies safe plans through typed policies supplied by
the consuming artifact family. It does not own domain identity, durable IDs,
warning semantics, or artifact construction rules.
Generic layers must not depend on production extensions. Concrete extensions
must not compose the application or take ownership of process behavior. The
current packages implementing these layers are inventoried in
@@ -183,6 +195,21 @@ The caller of the LLM owns prompt selection, prompt inputs, response schema,
and interpretation of structured output. Provider adapters do not own source-
or domain-specific prompt logic.
PromptKit owns bounded structural correction within one structured completion.
Notarius owns outer stage attempts, semantic validation, and acceptance policy;
the two budgets must remain separate.
When a model selects an application entity, callers must supply a contextual
selection and deterministically attach the opaque application identity whenever
the selection resolves exactly. Models do not receive or reproduce opaque
application identifiers. Semantic reconciliation may instead expose
contiguous, one-based candidate handles that exist only for one request;
deterministic code resolves them before typed application, and they never
become durable identity. This is the approved request-local-label application
of [ADR-0012](../adr/0012-resolve-opaque-entity-identifiers-deterministically.md)
recorded by
[ADR-0013](../adr/0013-use-request-local-candidate-handles-for-semantic-reconciliation.md).
LLM calls and other external operations accept cancellation and respect
timeouts. Concurrency control belongs in shared runtime plumbing rather than in
individual modules.
@@ -238,6 +265,16 @@ contain application data and therefore inherits its sensitivity; operators own
access controls and retention. Physical layout and operation are defined in
[Operations](../operations.md).
## Platform And Distribution
Linux is the supported deployment platform. macOS is supported only as a
best-effort development and compilation environment, while Windows is
unsupported. Notarius distributes source releases only: an immutable source
tag and its checked-in release note identify a release. The project does not
publish executable binaries, archives, installers, container images,
checksums, signatures, or package-manager entries. Maintainer release commands
and tag guards belong to [Source Releases](../release.md).
## Architectural Non-Goals
Notarius does not aim to provide:

View File

@@ -65,6 +65,8 @@ secret values.
| CLI contract | `docs/cli.md` | Commands, arguments, flags, invocation semantics, and exit codes. | End-to-end operating procedures, configuration field definitions, runtime filesystem layout, module implementation details. |
| Configuration contract | `docs/config.md` | Discovery and precedence, file schema, fields, defaults, environment overrides, validation rules, and user-selectable module or validator keys. | Complete example files, CLI syntax, runtime state lifecycle, module implementation details. |
| Operations | `docs/operations.md` | Runtime workflows, physical filesystem and state layout, output, cache, and debug handling, resume, cleanup, permissions, recovery, and operational limits. | CLI flag syntax, configuration field definitions, logical output schemas, implementation mechanics. |
| Source release procedure | `docs/release.md` | Maintainer release selection, candidate validation, tagging, publication guards, verification, and immutable-tag recovery. | Product installation summary, CLI version semantics, historical release summaries, CI implementation detail. |
| Release-note history | `docs/releases/` | One checked-in historical summary for each source release made under the procedure. The note at the immutable tag is that release's record. | Current commands, behavior, contracts, and compatibility definitions. |
| Public HTTP contract, if introduced | `docs/api.md` | Routes, authentication, media types, request and response schemas, status codes, pagination, caching, idempotency, rate limits, and HTTP retry semantics. | Client walkthroughs, upstream or downstream integration internals, implementation detail. |
| Consumer guidance, if a public package or API is introduced | `docs/consumers/` | Task-oriented use of the public interface, minimal client examples, and consumer responsibilities. | HTTP wire semantics, external protocol contracts, internal implementation detail. |
| External and durable integration contracts | `docs/integrations/` | External file formats and protocols, upstream and downstream contracts, logical output bundle paths and schemas, media types, and compatibility behavior. | Physical runtime placement and lifecycle, internal transformations, CLI syntax, configuration defaults. |
@@ -95,6 +97,14 @@ runtime state and how to operate or recover the application. When a workflow
crosses these topics, choose the document that owns the task and link to the
other contracts.
### Releases
`docs/release.md` owns the source-release procedure. Release notes are
historical summaries, not current-state contract owners: the checked-in note at
an immutable tag records that release, while current canonical documentation
must change with the behavior it describes. Do not use a release note to defer
or replace current documentation updates.
### Contracts And Implementation
Integration and API documents define externally observable shapes and

164
docs/release.md Normal file
View File

@@ -0,0 +1,164 @@
# Source Releases
This procedure is for maintainers publishing Notarius source releases. A
release is an immutable lightweight `vMAJOR.MINOR.PATCH` tag on `main` together
with its checked-in `docs/releases/<tag>.md` note. Tag CI validates that source
candidate after publication; it does not publish or repair a release.
Notarius publishes no binaries, archives, checksums, signatures, containers,
package-manager entries, or Gitea release objects. Windows is not supported.
Do not create retrospective notes for the pre-procedure `v0.1.0`, `v0.2.0`, or
`v0.3.0` tags.
## Select And Describe The Release
Choose an unused stable semantic version in the form `vMAJOR.MINOR.PATCH`.
Prereleases are not supported. Before `v1.0.0`, a minor release may change a
documented CLI, configuration, durable artifact, integration, or operating
contract when its note explains the impact and required operator action. A
patch release must not intentionally break those documented contracts within
its minor line.
Create the version-matched note as part of the candidate. Every new note uses
this structure, with concise, truthful content in each section:
```markdown
# Notarius vMAJOR.MINOR.PATCH
This release ...
## Summary
## Compatibility
## Upgrade
## Changes
```
The note is a historical summary. Link to current canonical documentation for
exact behavior, and update that documentation in the candidate rather than
using the note as a substitute.
## Prepare The Candidate
Set the selected release version and disable Go workspace use for every
candidate command:
```sh
RELEASE_VERSION=vMAJOR.MINOR.PATCH
export RELEASE_VERSION GOWORK=off
```
Run the shared source-candidate checks from the repository. They cover module
hygiene, tests, race tests, vet, builds, formatting, whitespace, maintained
configuration validation, and the Linux and Darwin command-build matrix:
```sh
./scripts/check-release-source.sh "$RELEASE_VERSION"
```
Before committing, manually follow every changed local Markdown link and
review the candidate for unintended files, generated output, credentials, or
other unrelated changes. Commit the release note and all affected current
documentation, then run the shared checker against that exact candidate. Push
the candidate commit to `main` only after it succeeds. Record the exact commit
only after that push:
```sh
RELEASE_COMMIT=$(git rev-parse 'HEAD^{commit}')
export RELEASE_COMMIT
```
For private-module installation, configure standard `GOPRIVATE` matching this
module and ordinary Git authentication for the hosting service before running
the verification below. The exact authentication mechanism belongs to the
maintainer environment; never record credentials or environment dumps in a
release note, command history, or repository file.
## Guard And Publish The Tag
Fetch current remote references, then run this guard without editing the
candidate. It requires `main`, a clean worktree and index, disabled workspace
use, a stable release version, the recorded and pushed commit, a matching note,
and unused local and remote tags:
```sh
git fetch origin main --tags
if ! printf '%s\n' "$RELEASE_VERSION" |
grep -E -x 'v(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)' >/dev/null
then
printf '%s\n' "invalid release version: $RELEASE_VERSION" >&2
exit 1
fi
test "$GOWORK" = off
test "$(git branch --show-current)" = main
test -z "$(git status --porcelain)"
test "$RELEASE_COMMIT" = "$(git rev-parse 'HEAD^{commit}')"
test "$RELEASE_COMMIT" = "$(git rev-parse 'origin/main^{commit}')"
test -s "docs/releases/$RELEASE_VERSION.md"
grep -F -x "# Notarius $RELEASE_VERSION" "docs/releases/$RELEASE_VERSION.md"
for heading in '## Summary' '## Compatibility' '## Upgrade' '## Changes'; do
grep -F -x "$heading" "docs/releases/$RELEASE_VERSION.md"
done
if git rev-parse -q --verify "refs/tags/$RELEASE_VERSION" >/dev/null; then
printf '%s\n' "local tag already exists: $RELEASE_VERSION" >&2
exit 1
fi
if git ls-remote --exit-code --tags origin "refs/tags/$RELEASE_VERSION" >/dev/null 2>&1; then
printf '%s\n' "remote tag already exists: $RELEASE_VERSION" >&2
exit 1
fi
```
Create an explicitly lightweight tag against the guarded commit, verify its
target, and push only that tag ref:
```sh
git -c tag.gpgSign=false tag "$RELEASE_VERSION" "$RELEASE_COMMIT"
test "$(git cat-file -t "$RELEASE_VERSION")" = commit
test "$(git rev-parse "$RELEASE_VERSION^{commit}")" = "$RELEASE_COMMIT"
git push origin "refs/tags/$RELEASE_VERSION:refs/tags/$RELEASE_VERSION"
```
Never use `git push --tags`, move a published tag, or delete a published tag.
## Verify The Published Release
Confirm that the remote tag still points at the guarded commit and that the
note is available from the tagged tree:
```sh
REMOTE_TAG_COMMIT=$(git ls-remote origin "refs/tags/$RELEASE_VERSION" | awk '{print $1}')
test "$REMOTE_TAG_COMMIT" = "$RELEASE_COMMIT"
git show "$RELEASE_VERSION:docs/releases/$RELEASE_VERSION.md" >/dev/null
```
Verify a fresh source installation and its diagnostic version. The temporary
directory confines the installed command to this check:
```sh
release_verification_dir=$(mktemp -d)
trap 'rm -rf "$release_verification_dir"' 0 HUP INT TERM
mkdir -p "$release_verification_dir/bin"
GOWORK=off GOBIN="$release_verification_dir/bin" go install \
"gitea.maximumdirect.net/eric/notarius/cmd/notarius@$RELEASE_VERSION"
test "$("$release_verification_dir/bin/notarius" --version)" = "notarius $RELEASE_VERSION"
```
An exact fresh checkout and `GOWORK=off go build ./cmd/notarius` is an
equivalent source verification when local installation policy requires it.
`notarius --version` is diagnostic only; downstream compatibility remains
defined by the published receipt and artifact contracts.
## Failure And Correction Policy
If candidate validation fails before publication, fix the candidate on `main`,
rerun the shared checker, and repeat the guards. An unpublished local tag may
be deleted after inspection.
If the remote tag or tag CI reveals a defect, leave the published tag intact.
Fix the defect on `main`, choose a new patch version, write a new matching
note, and repeat this procedure. Do not weaken tag immutability or add release
assets as a workaround.

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,224 @@
# D&D Subprocess Consumer Documentation
## Status
Completed. The target guide is `docs/consumers/dnd-pipeline.md`.
## Purpose
Provide one task-oriented guide for applications that run Notarius as a
subprocess to execute the maintained complete D&D pipeline and consume its
published artifacts. The initial concrete consumer is Narratio, but the guide
must describe the public Notarius workflow rather than depend on Narratio
internals.
The guide should make the safe integration path obvious without duplicating
the CLI, input, receipt, output-bundle, or individual artifact contracts that
already have canonical documentation.
## Current State
The public integration surface is documented accurately but is distributed
across several documents:
- `docs/consumers/subprocess.md` defines the generic subprocess workflow;
- `docs/cli.md` owns commands, flags, stream behavior, and exit statuses;
- `docs/integrations/seriatim.md` owns the accepted transcript input shape;
- `docs/integrations/run-result.md` owns the machine-readable successful-run
receipt;
- `docs/integrations/json-output.md` owns bundle discovery and logical files;
- the D&D integration documents own the individual lane payload contracts;
- `examples/dnd-complete.config.yml` is the maintained complete pipeline.
A consumer can reconstruct the full workflow from those documents, but there
is no D&D-focused guide that connects the maintained example to its input,
invocation, complete artifact inventory, discovery procedure, and downstream
acceptance decisions.
## Target Documentation Set
### Create `docs/consumers/dnd-pipeline.md`
This document should own the end-to-end consumer workflow for the maintained
complete D&D configuration. It should be useful to Narratio and to another
subprocess orchestrator with the same needs.
The guide should contain the following sections.
#### Prerequisites And Deployment Configuration
- Link to `examples/dnd-complete.config.yml` rather than embedding a second
complete configuration.
- Explain that a deployment must provide the configured PromptKit profile and
campaign reference files.
- Recommend absolute paths for a service or orchestrator deployment.
- Call out the path-resolution distinction explicitly: YAML reference paths
are relative to the Notarius configuration file, while
`promptkit.profile_file` is relative to the Notarius process working
directory.
- Recommend validating the selected configuration and `dnd-session` pipeline
before processing sessions.
#### Transcript Input
- State that the complete pipeline consumes a Seriatim JSON document.
- Link to the canonical Seriatim contract for required fields and validation.
- Recommend the caller's final trimmed transcript when the caller maintains
transcript tiers. For Narratio, identify the implemented source as
`narratio.transcript.final_trimmed`, normally stored at
`transcripts/final.trimmed.json`.
- Explain that segment IDs must remain stable because D&D source references
cite those units.
- Explain that Notarius derives its default prompt session from the input
module and exact input bytes and that ordinary callers should not supply
`--session-id`.
#### Subprocess Invocation
- Show one concise invocation using `notarius run dnd-session`, explicit
absolute `--config`, `--input`, and `--output-dir` paths, and `--json`.
- Direct callers to capture stdout and stderr separately, propagate
cancellation, impose an operator-appropriate timeout, and wait for process
completion before parsing stdout.
- State that only exit status zero permits receipt decoding and link to the CLI
contract for the complete exit-status definition.
- Recommend retaining stderr and the invocation context for diagnosis without
logging secrets or transcript content.
#### Receipt And Bundle Discovery
- Require callers to accept only supported run-result schema versions while
tolerating unknown fields allowed by that version.
- Direct callers to obtain the exact run-specific bundle from the receipt's
absolute `output_directory`; they must not scan for the newest run directory
or construct a run ID.
- Require a confinement check when resolving `index_file` beneath the reported
bundle root.
- Direct callers to discover lane payloads by `lane_id` in `index.json`, then
verify descriptor media type and schema identity before decoding them.
- Explain that descriptor paths are untrusted relative paths and require the
same confinement discipline.
#### Complete D&D Artifact Inventory
Include a compact table for the ten lane IDs selected by the maintained
complete configuration:
- `item-registry`;
- `npc-registry`;
- `location-registry`;
- `scene-descriptions`;
- `item-occurrences`;
- `spells`;
- `combat-turns`;
- `npc-occurrences`;
- `location-occurrences`;
- `enemy-events`.
For each row, give a one-line purpose and link to the corresponding canonical
D&D artifact contract. Do not copy its fields or schema rules into the
consumer guide.
Document the four always-published bundle files—`index.json`, `manifest.json`,
`rejected.json`, and `warnings.json`—and the complete example's configured
`chunk-map.json` and `evidence-context.json` pipeline-wide artifacts. Link to
their canonical contracts and distinguish pipeline-wide artifacts from lane
outputs.
The inventory must say that a file is available only when its corresponding
artifact was accepted and published. It must not imply that process success
guarantees every configured lane.
#### Downstream Acceptance And Retention
- Explain that exit status zero can coexist with rejected outputs, warnings,
or absent lane descriptors.
- Require the consumer to define its required lane set explicitly. Recommend
treating all ten lanes as required when the caller claims to consume the
complete D&D workflow, while allowing another consumer to adopt a narrower
documented policy.
- Recommend retaining the receipt, the complete published bundle, and captured
diagnostic streams long enough to support provenance and failure analysis.
- Explain that `evidence-context.json` is a reading excerpt; authoritative
citations remain in lane payloads.
- Treat transcripts, lane artifacts, evidence context, manifests, and logs as
sensitive campaign data.
#### Compatibility Checklist
End with a concise checklist covering process exit, receipt schema, path
confinement, pipeline identity, index decoding, required descriptors,
descriptor schema/media compatibility, warnings and rejections, checksums or
retention, and secure handling. Compatibility should be based on published
receipt and artifact contracts rather than parsing a human version string.
### Update Existing Navigation
- Add a short link from `docs/consumers/subprocess.md` to the D&D-specific
workflow. Keep generic subprocess policy in the existing document.
- Add the guide to the documentation links in `README.md`.
- Extend the subprocess-consumer row in `docs/development.md` so maintainers
working on the D&D workflow are routed to the new guide and the canonical
contracts.
### Verify Canonical Contract Documents
Review the linked integration documents and the complete example while writing
the guide. Correct an integration document only if repository inspection finds
an actual stale contract. Do not move schema definitions, field tables, CLI
flags, or configuration semantics into the new guide.
## Narratio Alignment
The guide may name Narratio as the motivating consumer and identify its current
final-trimmed transcript source. It must not claim that Narratio already has a
Notarius adapter or extraction stage. Until that feature is implemented,
Narratio-specific architecture, configuration, stage behavior, manifest
records, and artifact source IDs belong in Narratio's roadmap.
Once Narratio implements the integration, its own integration documentation
should link to this guide and the durable Notarius contracts instead of
repeating them.
## Validation
Documentation implementation should include:
```sh
go run ./cmd/notarius config validate \
--config examples/dnd-complete.config.yml \
--pipeline dnd-session
go test ./...
```
Also verify all new and changed relative Markdown links, compare the artifact
inventory directly with the maintained complete configuration, and confirm
that commands and path semantics match the CLI and configuration references.
If the repository still has no automated link checker, record that fact and
perform a focused manual link review.
## Acceptance Criteria
- A subprocess integrator can follow one D&D-focused guide from a Seriatim
transcript through safe discovery of every artifact configured by the
complete example.
- The guide makes stdout, stderr, exit-status, receipt, and path-confinement
responsibilities unambiguous.
- The ten configured D&D lanes and both configured pipeline-wide artifacts are
listed and linked to their canonical contracts.
- The guide distinguishes process success from the caller's required-artifact
policy.
- The profile-path and reference-path resolution rules are clearly stated.
- Existing navigation makes the guide discoverable.
- No volatile contract is defined in two places, and no unimplemented Narratio
behavior is presented as current.
## Non-Goals
- Implementing or documenting Narratio's future adapter or stage as current
Notarius behavior.
- Adding a new Notarius command, receipt version, output format, or artifact
schema.
- Duplicating the complete configuration or individual D&D payload schemas in
prose.
- Defining a universal partial-result policy for every Notarius consumer.

View File

@@ -5,13 +5,207 @@ configuration, operations, internal, and integration docs. This roadmap records
future work only. Items are ordered roughly by current value and specificity,
not as committed release dates.
## Near-Term Validation And LLM Reliability
The following work forms one related program but should be promoted into
separate feature roadmaps and implemented in dependency order. PromptKit owns
structural output repair within one completion. Notarius owns stage candidates,
validator chains, semantic rejection policy, and whether another stage attempt
is warranted.
### 1. Upgrade To PromptKit v0.8.0
This item has been promoted to the standalone
[PromptKit v0.8.0 Upgrade](promptkit-v0.8.md) roadmap. That document owns the
release-by-release compatibility review, adopted features, structured-repair
policy, target integration boundary, acceptance criteria, and settled design
decisions.
### 2. Feedback-Aware Stage Validation Retries
- Model Notarius's corrective stage-retry conversation explicitly after
PromptKit v0.8.0. The first attempt sends the ordinary complete initial
prompt. If application validation rejects the resulting LLM-produced
candidate and another stage attempt is available, reconstruct that complete
initial prompt byte-for-byte and append exactly two messages: an assistant
message containing the defective response and an application-owned user
message detailing every applicable semantic validation error and requesting
one corrected, complete replacement response. This is a freshly constructed
correction request, not continuation of an accumulating conversation.
- Use the configured stage `retries` value as the one outer retry budget for
this loop. `retries: N` continues to mean at most `N` additional complete
chunk, extract, merge, or normalize attempts after the initial attempt,
whether an attempt is needed because of a producer error or semantic
rejection. Do not add a second semantic-correction count. PromptKit's
prompt-level `repair_attempts` budget is independent and internal to each
individual LLM completion, and does not consume or replenish the Notarius
stage budget.
- Extend the framework-managed validation boundary for chunk, extract, merge,
and normalize stages so a rejected LLM-produced candidate and its exact raw
model response remain available to construct the next stage attempt.
Deterministic producers cannot improve by repeating the same inputs; a
rejection from a deterministic stage is therefore terminal under the
configured rejection policy rather than consuming retries mechanically.
- Preserve the original session ID, selected profile, structured-output
contract, prompt inputs, and reusable prompt prefix. Carry only the latest
candidate and latest aggregate feedback; do not build an unbounded retry
conversation. Keep model-facing corrective guidance separate from
operator-facing diagnostics, and apply explicit size, redaction, and debug
disclosure rules to both.
- Run every applicable validator in the configured chain before deciding
whether to retry. Do not short-circuit merely because an earlier validator
rejected the candidate. Aggregate all semantic rejection reason codes and
corrective guidance into the retry message so one retry can address the
whole candidate. A validator is applicable only when its declared target and
prerequisites can be satisfied; record a deterministic skipped diagnostic
rather than invoking a validator on an input it cannot interpret. Initially
execute the chain sequentially in configured order so results, diagnostics,
costs, and feedback ordering remain deterministic; consider validator
concurrency only in response to measured latency.
- Continue running independent applicable validators after one validator
execution failure so the attempt retains as much useful diagnostic
information as practical. Do not present validator operational failures as
defects in the producer candidate and do not include them in corrective
feedback.
- Distinguish three terminal conditions and make their policies configurable
at a coherent pipeline or binding scope:
- **producer structural failure:** PromptKit could not return a usable
structured candidate after its repair budget. Default to `fail_run`; an
allowed alternative may record a terminal stage or lane rejection where
execution can safely continue, but may not accept the invalid output;
- **semantic rejection:** one or more validators completed and rejected the
candidate. Default to `fail_run` after corrective stage retries are
exhausted; allow an explicit alternative that records the existing
rejected-output outcome without advancing that output;
- **validator execution failure:** a validator could not produce a valid
decision because of generation, structural-output, transport, or internal
failure. Default to a genuine warning and an explicitly recorded
`validation_incomplete` or equivalent degraded state while allowing the
candidate to continue; allow strict configuration to fail the run instead.
- An LLM-backed validator uses the same scheduled PromptKit boundary as every
other LLM-backed module. Its own response may use PromptKit's bounded
structural repair. Distinguish its possible output states:
- output rejected by PromptKit's structural contract should consume only the
validator prompt's configured PromptKit repair budget;
- output that is structurally valid but violates a deterministically
checkable validator-result invariant should be classified as a validator
execution failure;
- output that satisfies the complete validator-result contract is the
validator's decision, even though an LLM judgment may remain imperfect.
Automatically judging that judgment would require another semantic
validator and is outside this feature.
If the validator cannot return a contract-valid decision, do not recursively
create another Notarius semantic-validation loop around it. Apply the
configured validator-failure policy. The default warning must identify the
validator and affected stage without exposing sensitive content.
- Separate validator execution retry from producer correction. A transient
validator operational failure must not automatically discard and regenerate
an otherwise usable producer candidate. Any bounded retry of the validator
itself should reuse that same immutable candidate and remain subordinate to
PromptKit and provider retry behavior.
- Preserve attempt-level provenance, cumulative token usage, validator
outcomes, aggregated correction feedback, and terminal policy decisions in
the debug and manifest models without copying raw source material into
ordinary errors or durable summaries.
- Define terminal-outcome precedence. A semantic rejection dominates a
validator execution failure for the same candidate: use the completed
rejections to correct the producer while separately recording incomplete
validation. If a later candidate has no semantic rejection but one validator
still fails, apply the configured validator-failure policy to that candidate.
Never allow a known semantic rejection to become accepted through a
warn-and-continue setting, and never accept a structurally invalid producer
response. Permissive policy may preserve a rejected-output outcome or accept
a structurally valid candidate with explicitly incomplete validation; it may
not relabel known-invalid output as approved.
Before implementation, record the generic validation and retry state machine
in an ADR. The ADR should own the separation between PromptKit repair and
Notarius correction, use of the existing stage-retry budget, reconstruction of
correction conversations, all-applicable-validator aggregation, deterministic
validator ordering, non-recursive validator failure handling, outcome
precedence, default fail-open/fail-closed choices, configurable terminal
policies, and provenance and sensitive-data constraints. A dependency-upgrade
ADR is not needed for PromptKit v0.8.0 itself. Current behavior remains
authoritative until the validation ADR is implemented and the canonical
architecture, configuration, operations, and internal documentation are
updated.
### 3. D&D Combat Scene Semantic Validation
- Add an optional production LLM-backed D&D validator that determines whether
proposed scene boundaries and classifications represent substantive active
combat correctly. Its central quality goal is that active combat is kept in
coherent scenes classified as `combat`, rather than split incorrectly or
hidden inside scenes classified as `narrative`, `recap`, or `meta`.
- Resolve the validator's exact target before implementation. The current
`dnd/scenes` chunker owns only complete, gap-free source ranges, while the
per-chunk `dnd/scene-descriptions` extractor owns the `combat`, `narrative`,
`recap`, and `meta` classification. The preferred initial placement is
therefore an extract-stage validator for `dnd/scene-descriptions`, where it
can compare one proposed kind with the corresponding transcript chunk.
- Consider a chunk-stage LLM validator only for a distinct boundary-coherence
question that can be answered from the complete transcript and proposed
range map, such as whether one continuous combat was fragmented across
inappropriate scene boundaries. Do not duplicate the same classification
judgment at both stages. Moving classification into chunk-plan annotations
would change the deliberately minimal, annotation-free chunk contract and
requires an explicit architecture review before it is selected.
- Validate both false negatives and false positives: a non-combat kind must not
omit substantive active combat, and a combat kind must be supported by such
combat. Keep the existing deterministic downstream rule that combat-turn
extraction runs only for an exact `combat` scene classification; semantic
review improves the upstream classification but does not replace that gate.
- Run the semantic validator through PromptKit, use a minimal required-field
structured response schema, and let PromptKit repair structural validator
output within its bounded budget. A contract-invalid final validator response
is a validator execution failure, not a semantic rejection and not a reason
to recursively validate the validator.
- Evaluate the prompt and decision policy against a small human-reviewed set
containing combat setup, active turns, interruptions, multi-phase encounters,
brief rules discussion, aftermath, recalled combat, and false-positive
hostile dialogue. Measure false acceptance, false rejection, retry success,
added calls, latency, and token cost before placing it in the production
default chain.
- An ADR is not required if classification remains owned by
`dnd/scene-descriptions` and the validator follows the generic validation ADR.
Create or supersede an ADR if the work transfers scene classification into
the chunker or otherwise changes stage ownership or the durable chunk-plan
contract.
### 4. Warning Signal And Presentation Reform
- Audit every warning producer and representative successful runs. Ordinary
success producing dozens of warnings is a failed operator experience: the
volume obscures actionable problems and trains operators to ignore the
warning channel.
- Define a small warning taxonomy that distinguishes actionable degradation,
incomplete validation, lossy fallback, and data-quality risk from routine
normalization observations or informational diagnostics. Preserve detailed
traceability in debug or manifest data without promoting every observation
to a top-level CLI warning.
- Consider stable deduplication and aggregation by scope and reason code,
bounded samples plus omitted counts, and a concise CLI summary with a path to
detailed diagnostics. Do not suppress genuine validator execution failures
merely to reduce the count.
- Decide which warnings affect process status, rejection summaries, durable run
receipts, or only debug output. Ensure warning ordering and aggregation are
deterministic across concurrent execution.
- Establish a representative warning-volume acceptance target and human review
workflow before changing individual producers piecemeal. The intended result
is not zero warnings; it is a small set in which every surfaced warning merits
operator attention.
- This work does not require an ADR unless it changes validation acceptance,
failure, or durable contract semantics. CLI presentation and diagnostic
taxonomy otherwise belong in a feature roadmap followed by updates to their
canonical configuration, operations, integration, and internal documents.
## Near-Term D&D Pipeline
### Evaluate Spell Extraction And Normalization
- Evaluate ordinary extraction retries and the completed normalization path
against a human-reviewed transcript set before adding repair-aware retries or
an LLM-backed semantic validator.
against a human-reviewed transcript set before and after adopting the shared
PromptKit repair and Notarius validation-retry policies above.
- Maintain a small set of human-reviewed transcripts and outputs for prompt,
validator, and normalizer development. Treat model-quality review as an
iterative human evaluation aid, not a deterministic correctness gate.
@@ -24,26 +218,51 @@ not as committed release dates.
## Shared Normalization And Quality Work
### Generic LLM-Assisted Deduplication
The implemented source-backed core and initial D&D registry adoption are
described by [Module Internals](../internal/modules.md#semantic-reconciliation)
and
[D&D Module Internals](../internal/dnd.md#semantic-registry-reconciliation).
The sections below keep broader extensions deferred.
- Add a reusable normalizer that asks an LLM to identify duplicate sets in a
list and propose one replacement element for each set.
- Define the minimum domain-neutral input contract, initially an ordered list
whose elements have stable unique IDs. Artifact-kind registrations or
adapters may expose that structure without moving domain rules into the
generic package.
- Keep mutation deterministic: parse and validate the model's duplicate groups,
require every referenced ID to exist, reject overlapping or malformed groups,
prevent unrelated insertion or deletion, and apply only approved replacement
operations in code.
- Preserve provenance needed for audit and downstream validation, and emit
warnings describing every collapsed group.
- Evaluate batching and context-window limits before applying the normalizer to
large artifact collections.
### Large-Collection Semantic Reconciliation
The model may use its own domain knowledge to judge semantic duplication; the
generic implementation is responsible only for the common proposal contract,
safety checks, and deterministic application of accepted changes.
- Evaluate deterministic candidate blocking only after representative registry
inputs exceed the active roadmap's bounded single-request limits. Blocking
should use cheap, explainable signals to form plausible comparison sets while
preserving the possibility that a duplicate appears outside a lexical name
match.
- Define correctness for candidates that appear in more than one block,
conflicting canonical selections, transitive identity across blocks, retry
isolation, and deterministic final ordering before implementation.
- Prefer a reconciliation graph or union plan with explicit conflict checks
over arbitrary fixed-size slices. Never silently treat a batch boundary as
evidence that two candidates are distinct.
- Record per-request bounds, block provenance, model calls, discarded
proposals, and final group derivation well enough to audit a collapse.
### Operator-Selected Semantic Policies
- Consider allowing an operator to select an approved semantic-policy prompt
for a typed reconciliation module without replacing the shared protocol,
response schema, or deterministic safety rules.
- Define the trusted asset source, configuration syntax, compatibility checks,
startup validation, provenance, prompt fingerprinting, checkpoint effects,
and support boundary before exposing the option.
- Prefer selection among registered, typed-policy-compatible prompt assets over
arbitrary filesystem prompt paths. Do not add this flexibility until an
operator workflow requires it; artifact-family-owned policy remains simpler
and safer for the initial implementation.
### Broader Reconciliation Inputs And Module Selection
- Revisit alternate context providers when a concrete non-source-backed entity
collection needs semantic reconciliation. Any extension must preserve the
same request-local identity, deterministic proposal validation, provenance,
and typed application guarantees.
- Consider a selectable generic normalizer only if Notarius gains a real
domain-neutral typed artifact contract that can safely support it. Do not
weaken exact artifact registration or introduce reflection-based arbitrary
JSON mutation merely to expose a universal module key.
### Validation And Review
@@ -100,6 +319,18 @@ checkpoint reuse, when an older artifact may be decoded or adapted, and when a
producer or all dependents must be recomputed. Do not add a general migration
framework until an actual contract change requires one.
### Artifact-family-oriented physical packaging
[ADR-0004](../adr/0004-package-modules-by-domain.md) currently groups production
extensions by domain and then by pipeline stage. After artifact-family
ownership terminology is established and more families span extraction,
normalization, validation, codecs, references, and assets, reassess whether a
feature-first physical layout would improve navigation and reduce scattered
changes enough to justify a repository-wide package migration. Any change must
address Go dependency cycles, registrar ownership, stable public module keys,
and supersession of the affected ADR-0004 decision. Conceptual artifact-family
ownership does not by itself require this move.
## Blue-Sky Platform And Operations
These ideas are intentionally less specified. Promote one into an earlier
@@ -115,7 +346,6 @@ section only after a concrete workflow, contract, and priority emerge.
### Distribution And Operations
- Packaged release artifacts for alpha distribution.
- A documented versioning and release process.
- Optional generated example-output fixtures with a regeneration procedure.
- Additional diagnostics or reporting views.

View File

@@ -0,0 +1,782 @@
# PromptKit v0.8.0 Upgrade Implementation Plan
## Purpose
Implement the target state defined by the
[PromptKit v0.8.0 Upgrade](promptkit-v0.8.md): adopt the useful PromptKit
v0.6.0, v0.7.0, and v0.8.0 changes; enable one bounded structural correction
by default; expose pipeline and binding overrides; preserve safe provider
diagnostics; and keep PromptKit behind Notarius's transport-neutral LLM
boundary.
This plan is ordered. Each numbered stage is one implementation prompt for a
gpt-5.6-terra coding agent. Complete and validate one stage before beginning
the next. Read `docs/development.md` and every policy under `docs/policy/` at
the start of each stage, inspect the current code and tests named by that
stage, preserve unrelated worktree changes, and update current-behavior
documentation in the same stage as the behavior it describes.
Do not retire this plan or `promptkit-v0.8.md` during implementation. Keep both
until the completed work has passed a separate review. Do not implement the
future Notarius semantic-validation retry loop, D&D combat-scene validator, or
warning redesign as part of this plan.
## Decisions Fixed For Implementation
- Pin `gitea.maximumdirect.net/eric/promptkit` v0.8.0 directly, with no
`replace`, workspace dependency, or vendored source.
- Every maintained eligible production prompt defaults to exactly one
PromptKit structural repair attempt.
- Add the exact configuration key
`structured_output_repair_attempts` at pipeline scope and on LLM-backed
module and validator bindings.
- Effective precedence is binding value, then pipeline value, then the prompt's
declared `repair_attempts` value. Omission inherits; explicit zero disables
structural repair at that scope.
- Accepted values are integers from zero through three. Explicit null and
non-integer values are invalid. An explicit binding value on a deterministic
module or validator is invalid. A pipeline value is applied only to selected
LLM-backed bindings and does not make deterministic bindings invalid.
- Keep file configuration version 4. This is an additive pre-release field and
does not require parallel versioned behavior.
- Use `StructuredOutputRepairAttempts *int` for presence-aware internal Go
fields. Clone pointers at every ownership boundary.
- A configured override never replaces schema identity, output format, or
validation mode. The PromptKit adapter calls `InspectPrompt`, copies the
complete normalized prompt-owned output contract, changes only
`RepairAttempts`, and supplies the complete replacement on `RunRequest`.
Do not add an inspection cache initially.
- PromptKit repair is internal to one `CompleteStructured` call and does not
consume or replenish a binding's existing `retries` budget.
- Add `RepairAttempts int` to Notarius's structured-completion response. It is
the actual corrective-call count reported by PromptKit; token usage remains
PromptKit's cumulative usage and must not be summed again.
- A valid repaired response is successful and produces no warning solely
because repair occurred. Exhausted structural validation maps to
`ErrInvalidStructuredOutput` with the final candidate and debug material
retained.
- Add an application-owned generation-error sentinel and typed status-bearing
error. PromptKit error types must not cross `internal/framework/llm`.
- HTTP status may appear in the application-owned generation error. Provider
code, type, and message are excluded from ordinary errors, warnings,
manifests, cache, and checkpoint identity; they may appear only in an
explicitly requested debug trace after Notarius redaction.
- Profile inheritance is owned entirely by PromptKit. Notarius passes sources
through, inspects and records the resolved target, and does not parse or merge
`base_profile` itself.
- PromptKit's built-in `rakestrawhome` backend and
`rakestrawhome-gemma-4-31b` profile are available generically. Notarius does
not register, shadow, or select them by default.
- Missing optional credential environment values are allowed to reach the
provider without `Authorization`; Notarius does not recreate v0.5.0's local
failure or add provider-specific authentication logic.
- No dependency-upgrade ADR is required. Update architecture only with the
durable ownership distinction between PromptKit structural repair and
Notarius stage/semantic validation policy.
## Stage 1: Upgrade The Dependency And Establish A Clean v0.8.0 Baseline ✅
### Goal
Move the repository to PromptKit v0.8.0, resolve source-compatibility issues,
and establish a passing baseline before adopting new behavior.
### Implementation
1. Re-read the upstream v0.6.0, v0.7.0, and v0.8.0 release guides and the
v0.8.0 package consumer and format documentation. Treat the pinned v0.8.0
tag, not the sibling checkout's moving branch, as authoritative.
2. Update `go.mod` and `go.sum` to PromptKit v0.8.0 and run `go mod tidy` with
`GOWORK=off`.
3. Compile before making compatibility edits. Correct only actual source or
behavior incompatibilities. In particular:
- convert any positional `promptkit.Profile` or
`promptkit.OpenAICompatibleProfileConfig` literals to keyed literals;
- confirm Notarius does not register the newly reserved `rakestrawhome`
backend ID; and
- preserve `PrepareExecution`/`Details`/`RunPrepared` snapshot ownership,
`Discard`, session forwarding, reasoning override, profile preflight,
and capacity adaptation.
4. Change `promptKitBuiltinProfileCatalogID` in
`internal/framework/llm/promptkit_profile_fingerprint.go` from the v0.5.0
catalog marker to an opaque v0.8.0 marker. Do not hash PromptKit internal
files or include catalog content in manifests.
5. Update `docs/integrations/pkg-promptkit.md` to pin and link v0.8.0 and to
state that this stage still leaves the production prompt-declared repair
budget at its current value. Do not document later configuration or default
behavior before it exists.
6. Update only those existing tests whose public PromptKit types or stable
v0.8.0 behavior genuinely changed. Do not rewrite tests merely to match
upstream diagnostic wording.
### Tests And Validation
```sh
GOWORK=off go mod tidy -diff
GOWORK=off go test ./internal/framework/llm ./internal/cli
GOWORK=off go test ./...
GOWORK=off go vet ./...
GOWORK=off go build ./cmd/notarius
git diff --check
```
### Acceptance Criteria
- `go list -m gitea.maximumdirect.net/eric/promptkit` reports v0.8.0.
- There is no PromptKit `replace`, active Go workspace dependency, or vendor
tree.
- The adapter still uses one frozen prepared execution and all existing LLM
tests pass.
- Checkpoint profile identity includes the v0.8.0 built-in catalog marker.
- Current integration documentation pins v0.8.0 without claiming that
later stages are already active.
- The full ordinary test suite, vet, and command build pass.
## Stage 2: Verify v0.6.0 Compatibility And Hardening ✅
### Goal
Audit Notarius's assets and boundary values against PromptKit v0.6.0's stricter
source, path, endpoint, JSON, and cancellation contracts, fixing only concrete
incompatibilities.
### Implementation
1. Inspect `internal/framework/llm/asset_registry.go`, prompt/profile source
composition, all registered asset roots, the conventional local backend,
and their focused tests.
2. Exercise every production asset registry through PromptKit engine
construction and the existing production composition tests. Confirm that:
- YAML IDs and versions, not filenames, select definitions;
- every `content_file` path is exact, relative, contained, and points to a
regular embedded file;
- every schema and JSON asset is one complete JSON value;
- every current output contract is valid under v0.8.0; and
- unrelated malformed definitions do not create a second Notarius identity
or fallback mechanism.
3. Review local endpoint parsing and validation. Retain a narrower Notarius
rule only if it has independent application value; otherwise rely on
PromptKit's absolute HTTP/HTTPS URL contract. Never accept a value that the
adapter will later reject.
4. Review conversion of Notarius variables, inputs, profile extras, and debug
values at the adapter boundary for PromptKit's bounded JSON-compatible-value
rules. Do not add a second generic JSON walker or duplicate upstream numeric
limits.
5. Verify cancellation and deadline identity through existing adapter tests.
Add or refine one focused regression only if Notarius currently destroys an
`errors.Is`-relevant context or transport error that the application owns.
6. Do not add a cross-operation schema cache, artifact cache, provider-body
reader, or duplicate JSON framing validation; v0.6.0 owns those mechanisms.
### Tests And Validation
Run the focused asset, profile-source, and adapter packages, then the ordinary
and race-enabled suites:
```sh
GOWORK=off go test ./internal/framework/llm ./internal/cli
GOWORK=off go test ./...
GOWORK=off go test -race ./...
git diff --check
```
Tests must remain offline and should validate Notarius's assembled boundary,
not reproduce PromptKit's internal path, JSON-depth, or response-size matrices.
### Acceptance Criteria
- Every maintained embedded prompt, schema, and fallback profile can be loaded
through the assembled v0.8.0 engine.
- Current local endpoint and JSON-compatible values either satisfy the stricter
upstream contract or fail during preparation with safe diagnostics.
- No duplicate PromptKit-owned cache, JSON, or response-bound mechanism is
introduced.
- Cancellation and deadline behavior remains discoverable at the Notarius
boundary.
- Ordinary and race-enabled tests pass.
## Stage 3: Adopt Profile Inheritance, Rakestrawhome, And Optional Credentials ✅
### Goal
Make the useful PromptKit v0.7.0 profile and backend behavior work through
Notarius's existing generic profile boundary without adding provider-specific
composition logic.
### Implementation
1. Inspect `promptkit_profiles.go`, `asset_registry.go`, profile fingerprinting,
CLI profile preflight, profile provenance recording, and their tests before
editing.
2. Add an offline integration test using a temporary operator profile source
whose leaf uses `base_profile`. Prove that:
- preflight reports the leaf ID;
- the effective backend, model, reasoning, and other inherited values match
the resolved PromptKit target;
- execution uses the same resolved target as inspection; and
- a missing parent or cycle fails before provider generation with a safe
profile-load diagnostic.
Do not duplicate PromptKit's entire field-by-field merge test matrix.
3. Add a checkpoint-safety test showing that changing a parent definition in
an operator profile directory changes Notarius's profile-source fingerprint
while profile content and paths remain absent from the fingerprint value.
Retain the v0.8.0 catalog marker as coverage for built-in-parent changes.
4. Verify `rakestrawhome-gemma-4-31b` through the ordinary profile inspector.
Assert its selected backend reaches Notarius's application-owned inspection
and provenance fields. Use a fake PromptKit client or transport if execution
coverage is needed; never contact the live service or require credentials.
5. Verify that Notarius registers no `rakestrawhome` override and that the
existing `local` registration remains independent.
6. Add one `httptest`-backed adapter integration test for a filesystem profile
with a missing optional `api_key_env`. The request must reach the test server
without an `Authorization` header. Add a focused in-memory PromptKit profile
test for `APIKeyRequired` only if needed to prove Notarius preserves upstream
preflight behavior; do not expose a new operator profile API.
7. Keep `assets/dnd/profiles/dnd-extraction.yaml` standalone and unchanged. No
matching v0.8.0 built-in profile owns its `openai/gpt-5.6-luna` target.
8. Update the current profile-source, deployment, and pinned-integration
sections in `docs/config.md`, `docs/operations.md`,
`docs/internal/llm.md`, and `docs/integrations/pkg-promptkit.md`. Link to the
pinned PromptKit format rules for inheritance. Explain that filesystem
profiles cannot express PromptKit's in-memory `APIKeyRequired` field and
that an optional missing credential may result in a provider 401/403.
### Tests And Validation
```sh
GOWORK=off go test ./internal/framework/llm ./internal/core/config ./internal/cli
GOWORK=off go test ./...
GOWORK=off go test -race ./internal/framework/llm ./internal/cli
git diff --check
```
### Acceptance Criteria
- Inherited operator profiles resolve identically during preflight and
execution, with the leaf ID and effective target kept distinct.
- Parent changes invalidate checkpoint reuse without leaking profile content or
paths.
- Rakestrawhome is available through generic PromptKit profile handling and is
not selected by default or registered by Notarius.
- Missing optional credentials omit authorization and reach the controlled
test provider; explicitly required credentials retain upstream behavior.
- Current documentation accurately describes the implemented profile and
credential behavior without duplicating PromptKit's merge algorithm.
## Stage 4: Adapt Structured Generation Errors Safely ✅
### Goal
Use PromptKit v0.7.0's structured generation errors for stable status
classification and debug-only provider diagnostics without leaking PromptKit
types or sensitive provider text.
### Implementation
1. In `internal/framework/contracts`, add:
- `ErrLLMGeneration` as the provider-neutral generation-failure sentinel;
- an application-owned `LLMGenerationError` with private status and safe
diagnostic fields, `Error`, `Unwrap`, and `StatusCode` methods; and
- a constructor that accepts a nonnegative status and an already-redacted
diagnostic. Status zero means no HTTP status was available.
Ordinary callers may inspect status with `errors.As` and category with
`errors.Is`, but cannot obtain provider code, type, or message from the
error.
2. Add an application-owned `LLMDebugProviderError` with `status_code`, `code`,
`type`, and `message` fields, referenced optionally from
`LLMDebugResponse`. This is debug material, not a manifest or durable public
artifact contract.
3. In `PromptKitClient.CompleteStructured`, preserve precedence in this order:
caller context cancellation/deadline, PromptKit capacity error, structured
PromptKit generation error, then other PromptKit generation failures.
Map every generation failure to `ErrLLMGeneration`; map
`*promptkit.GenerationError` to `LLMGenerationError` with its status.
Never wrap or return the PromptKit error value itself.
4. Keep the ordinary diagnostic limited to PromptKit's safe default error
formatting after bearer and known-credential redaction. Do not append
`ProviderCode`, `ProviderType`, or `ProviderMessage` to it.
5. For an explicitly requested debug path, preserve prepared prompt details and
attach the PromptKit provider code, type, and message after:
- reading only the selected prepared target's `APIKeyEnv`, if any, to obtain
the exact known credential solely for redaction;
- applying `RedactSecrets` and the existing bearer/key-pattern redaction;
- retaining PromptKit's already-normalized bounds; and
- discarding the credential value immediately rather than storing it.
Do not scan unrelated environment variables.
6. Return prompt/debug material alongside the error so the existing debug LLM
wrapper can persist it only when debug recording is enabled. Confirm that
provider fields do not appear in ordinary error text, warnings, manifests,
cache, checkpoint data, or a run without debug output.
7. Refactor error mapping into small helpers if needed to keep
`CompleteStructured` readable; do not create provider-specific policy in
modules or the pipeline runner.
8. Update the error and observability sections of `docs/internal/llm.md` and
`docs/integrations/pkg-promptkit.md`. Keep operator disclosure rules in
`docs/operations.md` concise and link to the internal boundary where useful.
### Tests And Validation
- Use `httptest.Server` to return representative structured 400 and 503
responses. Assert `errors.Is(ErrLLMGeneration)`, `errors.As` to the
application-owned type, and the exact status without asserting complete
human wording.
- Include a provider message containing the selected test credential and a
bearer-shaped value. Verify both are absent from the ordinary error and
debug artifact, while a non-sensitive marker appears only in the requested
debug trace.
- Retain existing capacity and context tests to prove their more specific
classifications still win.
```sh
GOWORK=off go test ./internal/framework/contracts ./internal/framework/llm ./internal/framework/pipeline ./internal/cli
GOWORK=off go test ./...
GOWORK=off go test -race ./internal/framework/llm ./internal/framework/pipeline
git diff --check
```
### Acceptance Criteria
- PromptKit generation errors never escape the adapter error chain.
- All generation failures match `ErrLLMGeneration`; structured non-success
responses expose only application-owned HTTP status to ordinary callers.
- Provider code, type, and message are available only in an explicitly
requested, redacted debug trace.
- Capacity and context classifications remain unchanged and more specific.
- Security tests prove selected credentials and bearer tokens are not leaked.
## Stage 5: Add Adapter-Level Structured Repair Support ✅
### Goal
Teach the transport-neutral completion boundary and PromptKit adapter to apply
an optional repair override and report actual repair behavior, without yet
exposing the setting in pipeline configuration.
### Implementation
1. Add `StructuredOutputRepairAttempts *int` to
`contracts.StructuredCompletionRequest`. Copy the pointed-to value wherever
requests are cloned or retained.
2. Add `RepairAttempts int` to `contracts.StructuredCompletionResponse`. It is
the actual number of corrective generation calls, not the configured budget
and not the number of total candidates.
3. Validate a non-nil request value as zero through three at the adapter
boundary so programmatic callers cannot bypass later file/config validation.
4. When the request value is nil, leave `promptkit.RunRequest.Validation` nil
so the prompt's complete contract remains authoritative.
5. When the value is non-nil:
- call `Engine.InspectPrompt(ctx, promptID, promptVersion)`;
- copy `PromptInspection.OutputContract` by value;
- replace only `RepairAttempts`;
- pass the complete copied contract as `RunRequest.Validation`; and
- prepare and execute exactly as before.
Do not infer or hard-code schema paths, validation modes, or formats. Do not
cache inspection in this stage.
6. Map `result.Validation.RepairAttempts` to the response and leave
`result.Usage` cumulative values unchanged. The existing debug validation
object and prepared output contract should show actual and configured values
respectively.
7. Preserve result semantics:
- valid initial and repaired candidates decode normally;
- repair exhaustion returns the final raw candidate/debug material with an
error matching `ErrInvalidStructuredOutput`;
- explicit empty or whitespace-only content follows PromptKit validation;
- missing/null/non-string content remains a generation/provider failure;
- corrective-call generation errors use Stage 4's application-owned mapping;
and
- context cancellation wins at every error boundary.
8. Keep `CompleteStructured` and its helpers provider neutral outside this
adapter package. Do not expose PromptKit validation or inspection types.
9. Update only the adapter-owned repair behavior in `docs/internal/llm.md` and
`docs/integrations/pkg-promptkit.md`. State that public pipeline configuration
and the production default are added by later stages of this plan.
### Tests And Validation
Add adapter-level behavioral tests using a deterministic fake PromptKit LLM:
- nil override uses the prompt declaration;
- explicit zero overrides a positive prompt declaration without dropping its
JSON Schema contract;
- explicit one turns an invalid first candidate followed by a valid candidate
into one successful response with the final raw bytes, actual repair count
one, and cumulative usage;
- repair exhaustion returns the final candidate and validation diagnostics as
`ErrInvalidStructuredOutput`;
- explicit empty content is eligible for repair;
- a corrective generation failure maps through Stage 4; and
- invalid direct values below zero or above three fail before provider work.
Do not assert PromptKit's exact assistant/user correction prose or copy its
full internal repair matrix.
```sh
GOWORK=off go test ./internal/framework/contracts ./internal/framework/llm ./internal/framework/pipeline
GOWORK=off go test -race ./internal/framework/llm ./internal/framework/pipeline
GOWORK=off go test ./...
git diff --check
```
### Acceptance Criteria
- The adapter changes only repair count when applying a request override.
- Nil and explicit zero remain distinct.
- Repaired success returns final raw output, cumulative usage, and actual count
without a warning.
- Exhaustion, empty content, corrective generation failure, and cancellation
match the target semantics.
- No PromptKit type crosses the LLM package boundary.
## Stage 6: Propagate Repair Policy Through Framework Requests ✅
### Goal
Carry an optional effective repair budget from each resolved stage or validator
binding to its module request without changing public file configuration yet.
### Implementation
1. Add `StructuredOutputRepairAttempts *int` alongside `LLMProfile` to every
stage request that can belong to an LLM-backed binding:
- `ParseRequest`;
- `ChunkRequest`;
- `TypedExtractionRequest`;
- `TypedMergeRequest`;
- `TypedNormalizeRequest`;
- `OutputRequest`;
- `TypedValidationRequest`;
- `ChunkValidationRequest`; and
- `SerializedValidationRequest`.
2. Add the same optional field to the erased/internal request carriers used by
registry builders, preparation, runner stage attempts, validator targets,
retry closures, and debug wrappers. Copy pointer values; never share a
mutable pointer owned by configuration.
3. At every runner stage invocation, obtain the value from the exact resolved
producer binding. At every validator invocation, obtain it from that exact
resolved validator binding. Do not use the producer's value for a validator
or vice versa.
4. Ensure all retry attempts for the same binding receive the same effective
structural-repair value. Do not decrement it in Notarius; PromptKit owns the
inner budget independently on each `CompleteStructured` call.
5. Extend `semanticreconcile.Request` with the optional field and carry it into
each generic reconciliation completion. A batched reconciliation may make
several completion calls; each call receives the same effective budget.
6. Update registry erasure/adaptation code for typed merge, normalize, and
validation requests so no field is lost. Preserve input/output support even
though current production input and output modules are deterministic.
7. Add focused framework tests for one chunk producer, one extraction
producer, one normalizer, and one LLM-backed validator. Verify exact pointer
value propagation and separation between producer and validator settings.
Do not add repetitive tests for every generic adapter.
### Tests And Validation
```sh
GOWORK=off go test ./internal/framework/contracts ./internal/framework/pipeline ./internal/framework/semanticreconcile
GOWORK=off go test -race ./internal/framework/pipeline ./internal/framework/semanticreconcile
GOWORK=off go test ./...
git diff --check
```
### Acceptance Criteria
- Every stage and validator request can carry a detached optional repair value.
- The runner sources the value from the exact resolved binding.
- Producer and validator values cannot overwrite one another.
- Stage retries reuse but do not mutate or consume the inner repair budget.
- Semantic reconciliation forwards the budget to every one of its completion
calls.
- Existing behavior remains unchanged while all values are nil.
## Stage 7: Forward Repair Policy From Every LLM-Backed Module ✅
### Goal
Complete the internal end-to-end path by having every production LLM-backed
module forward its stage request value to `CompleteStructured`.
### Implementation
1. Inventory every production `CompleteStructured` call with code search before
editing. The expected current owners include:
- `dnd/scenes` chunking;
- the combat-turn, enemy-event, item-occurrence, item-registry,
location-occurrence, location-registry, NPC-occurrence, NPC-registry,
scene-description, and spell extractors; and
- generic semantic reconciliation used by the item, location, and NPC
registry normalizers.
Reconcile this list with the actual repository; do not omit a newly added
production caller merely because it is not named here.
2. In each direct caller, set
`StructuredCompletionRequest.StructuredOutputRepairAttempts` from the
corresponding stage request. Clone the pointer or use a small shared helper
if that reduces repeated ownership mistakes without moving domain logic.
3. Ensure D&D registry normalizers pass their typed normalize request value into
`semanticreconcile.Request`, and that the generic engine forwards it as
established in Stage 6.
4. Update existing module prompt-mapping tests that already inspect a captured
structured-completion request to assert the new field. Do not create a new
one-test-per-module suite solely to memorialize field plumbing; rely on the
existing request-contract tests plus a final complete call-site audit.
5. Search again after editing for production `CompleteStructured` calls and
verify each either forwards the field or documents why it cannot receive a
pipeline binding. Test-only fakes need only preserve the field when their
contract test depends on it.
6. Do not set a module-specific fallback value. Nil must reach the adapter so
the prompt declaration remains authoritative.
### Tests And Validation
Run focused D&D and semantic-reconciliation packages, then the full suite:
```sh
GOWORK=off go test ./internal/modules/dnd/... ./internal/framework/semanticreconcile
GOWORK=off go test -race ./internal/modules/dnd/... ./internal/framework/semanticreconcile
GOWORK=off go test ./...
git diff --check
```
### Acceptance Criteria
- Every production LLM-backed completion receives the exact stage or validator
repair value.
- No module invents a default or imports PromptKit.
- Registry normalizers preserve the value through semantic reconciliation.
- Existing request-contract tests remain concise and pass.
- A final call-site audit finds no silent production omission.
## Stage 8: Add The Public Repair Configuration Contract ✅
### Goal
Add presence-aware pipeline and binding configuration for
`structured_output_repair_attempts` without yet changing runtime resolution.
### Implementation
1. Add `StructuredOutputRepairAttempts *int` to
`pipeline.PipelineProfile` and `pipeline.ModuleBinding`, using
`json:"structured_output_repair_attempts,omitempty"`.
2. Add presence-aware YAML support at pipeline and object-binding scope:
- exact key `structured_output_repair_attempts`;
- integer values zero through three;
- explicit null, non-integer, and out-of-range values rejected with scoped
diagnostics; and
- scalar shorthand bindings continue to omit the binding override.
Preserve file configuration version 4.
3. Update every configuration clone, conversion, redaction, summary, and JSON
round-trip carrier. Copy pointers by value into newly allocated storage so
parsed, configured, and redacted values do not alias.
4. Preserve omission versus explicit zero through YAML parsing, profile
inheritance, module-binding object form, and JSON round trips. Keep scalar
shorthand bindings equivalent to omission.
5. Do not add a top-level `promptkit.repair_attempts` setting or CLI override.
6. Add concise parser and ownership tests. Defer execution-class checks,
effective precedence, resolved digests, and runtime forwarding to Stage 9,
where module metadata is available.
### Tests And Validation
At the parser/config boundary, test omitted, explicit zero, positive bounds,
negative, above-three, null, non-integer, scalar shorthand, cloning, redaction,
and JSON round-trip behavior. Use relational boundary tests for the allowed
range and avoid duplicating the same cases at every layer.
```sh
GOWORK=off go test ./internal/core/config ./internal/cli
GOWORK=off go test -race ./internal/core/config
GOWORK=off go test ./...
go run ./cmd/notarius config validate \
--config examples/dnd-minimal.config.yml \
--pipeline dnd-session
go run ./cmd/notarius config validate \
--config examples/dnd-complete.config.yml \
--pipeline dnd-session
git diff --check
```
### Acceptance Criteria
- The exact public field parses at pipeline and object-binding scope with the
fixed range.
- Nil and explicit zero remain distinguishable through parsing, cloning,
inheritance, redaction, summaries, and round trips.
- Both maintained configurations remain valid without requiring the new field.
- No runtime or prompt default has changed prematurely.
## Stage 9: Resolve And Apply Repair Configuration ✅
### Goal
Resolve the public field against module execution classes, incorporate the
effective value into pipeline identity, and connect it to the request plumbing
completed in Stages 6 and 7.
### Implementation
1. During resolution, compute the effective value for every selected binding:
- explicit binding value wins;
- otherwise an explicit pipeline value applies to an LLM-backed binding;
- otherwise leave nil for prompt-owned policy.
Apply the pipeline value to LLM-backed validators as well as producers.
2. Reject an explicit binding value on a deterministic module or deterministic
validator using the same execution-class knowledge used for `llm_profile`.
Do not reject a pipeline-level value merely because a selected pipeline also
contains deterministic bindings; simply do not apply it to those bindings.
3. Clone every resolved pointer so the parsed pipeline, resolved profile,
redacted summaries, and runner requests have distinct ownership.
4. Include the effective field in resolved pipeline JSON and digest input. A
change between nil, zero, and a positive value must change the resolved
digest when it changes an LLM-backed selected binding. Unselected lanes must
retain the repository's existing digest and selection semantics.
5. Pass the resolved value into the Stage 6 request field for every selected
input, chunk, extract, merge, normalize, output, and validator binding.
6. Update `docs/config.md` as the canonical field, range, and precedence
contract; `docs/internal/pipeline.md` as the resolution owner; and
`docs/operations.md` for the distinction from binding `retries`. The
prompt-owned production default remains unchanged until Stage 10.
7. Add focused resolution and runner tests. Cover representative execution
classes rather than repeating the same assertion for every module type.
### Tests And Validation
Test:
- binding over pipeline over nil precedence;
- inheritance into each selected LLM-backed stage and validator;
- no inheritance into deterministic bindings;
- explicit deterministic-binding rejection;
- detached pointers;
- runner forwarding for representative producer and validator bindings; and
- digest changes for execution-relevant nil, zero, and positive changes.
```sh
GOWORK=off go test ./internal/framework/pipeline ./internal/cli
GOWORK=off go test -race ./internal/framework/pipeline
GOWORK=off go test ./...
go run ./cmd/notarius config validate \
--config examples/dnd-minimal.config.yml \
--pipeline dnd-session
go run ./cmd/notarius config validate \
--config examples/dnd-complete.config.yml \
--pipeline dnd-session
git diff --check
```
### Acceptance Criteria
- The exact public field has the fixed binding-over-pipeline-over-prompt
precedence for every selected LLM-backed producer and validator.
- Deterministic binding misuse fails during resolution before execution, while
a pipeline value coexists with deterministic bindings.
- Nil and explicit zero remain distinguishable through resolution, runtime,
summaries, and digests.
- A policy change invalidates checkpoint identity when it changes an effective
selected binding.
- Current configuration, pipeline, and operations documentation matches the
implemented behavior.
## Stage 10: Enable The Default, Finish Documentation, And Verify The Feature ✅
### Goal
Set the accepted production default of one repair, reconcile all canonical
documentation, and run the full repository verification pass.
### Implementation
1. Change `repair_attempts: 0` to `repair_attempts: 1` in every maintained
production prompt manifest that produces structured output, including the
generic semantic-reconciliation prompt and every D&D chunk, extraction, and
registry-normalization prompt. Do not mechanically change unrelated test
fixtures whose purpose is to exercise zero.
2. Inspect every production prompt output contract after the edit. Confirm that
each positive budget uses `basic`, `json`, or `json_schema`, remains no
greater than three, and retains its existing format and schema path.
3. Add or refine the smallest durable assembled-assets test that proves the
production engine can prepare the maintained prompts with the activated
contracts. Do not add a brittle test that asserts an exact prompt count,
file count, message prose, correction text, or asset length. The public
default may be tested at one canonical assembled boundary because its
literal value is an operational contract.
4. Confirm a successful repair does not create a warning and that exhausted
repair remains `ErrInvalidStructuredOutput`. Verify the debug prompt records
the configured contract, the debug response records actual repair count,
and cumulative usage is not double-counted.
5. Confirm scheduling behavior with one focused test or existing coverage: the
Notarius scheduled client admits one logical `CompleteStructured` operation
while PromptKit may make serial corrective provider calls inside it. Do not
attempt to reacquire a Notarius permit from inside PromptKit or add a second
scheduler.
6. Reconcile current-state documentation:
- `docs/integrations/pkg-promptkit.md` owns the pinned upstream boundary;
- `docs/config.md` owns field names, range, default, and precedence;
- `docs/operations.md` owns latency/cost, optional credentials, concurrency,
timeout, and the upper-bound formula;
- `docs/internal/llm.md` owns inspection-based contract replacement,
cumulative usage, actual repair count, generation errors, and debug data;
- `docs/internal/pipeline.md` owns effective policy propagation and the
separation from stage retries; and
- `docs/policy/architecture.md` adds only the durable rule that PromptKit
owns deterministic structural repair inside one completion while Notarius
owns stage attempts and semantic validation.
7. Remove current-behavior claims that PromptKit is v0.5.0, that every
production repair budget is zero, or that PromptKit is always single-pass.
Do not alter historical release notes or archived roadmaps.
8. Keep the maintained minimal and complete examples secret-free and valid.
They may omit the new field to demonstrate the default; do not add a
redundant complete profile or Rakestrawhome example merely to exercise an
upstream catalog entry.
9. Review `docs/roadmap/future.md` only for consistency. Leave the future
feedback-aware stage retry, combat-scene validator, and warning-reform work
unimplemented and clearly separate.
### Tests And Validation
Run focused tests first, then all repository checks:
```sh
GOWORK=off go test ./internal/framework/llm ./internal/framework/pipeline ./internal/framework/semanticreconcile ./internal/modules/dnd/...
GOWORK=off go test ./...
GOWORK=off go test -race ./...
GOWORK=off go vet ./...
GOWORK=off go build ./cmd/notarius
GOWORK=off go mod tidy -diff
go run ./cmd/notarius config validate \
--config examples/dnd-minimal.config.yml \
--pipeline dnd-session
go run ./cmd/notarius config validate \
--config examples/dnd-complete.config.yml \
--pipeline dnd-session
git diff --check
```
Also perform focused repository searches that exclude `docs/roadmap/archive/`
and historical release notes:
- no active v0.5.0 PromptKit pins or links remain;
- no maintained production prompt still declares `repair_attempts: 0`;
- every production `CompleteStructured` caller forwards the repair field; and
- no provider code, type, or message is added to ordinary errors, warnings,
manifests, cache, or checkpoint schemas.
If the repository's source-release checker is available and the ordinary
checks above pass, run `./scripts/check-release-source.sh v0.0.0` as the final
integrated validation. It must not create a tag, release note, or repository
artifact.
### Acceptance Criteria
- Every maintained structured prompt defaults to one corrective call and can
be overridden to zero through three at pipeline or binding scope.
- A real assembled Notarius completion follows the PromptKit v0.8.0 repair
contract without changing prompt schema identity or cacheable prefix.
- Actual repair count, cumulative usage, error classification, debug-only
provider diagnostics, scheduling, and checkpoint identity match the feature
roadmap.
- Profile inheritance, Rakestrawhome availability, optional credentials, and
v0.6.0 hardening remain covered and documented.
- All canonical documentation describes implemented v0.8.0 behavior in its
assigned home and leaves future semantic validation work in the roadmap.
- Maintained examples validate, all ordinary/race/vet/build/module checks pass,
and the worktree contains no generated or sensitive artifacts.

View File

@@ -0,0 +1,520 @@
# PromptKit v0.8.0 Upgrade
## Status
Proposed.
## Purpose
Upgrade Notarius from PromptKit v0.5.0 to v0.8.0 and deliberately adopt the
useful correctness, profile-composition, provider-diagnostic, backend, and
structured-output-repair capabilities introduced in PromptKit v0.6.0, v0.7.0,
and v0.8.0.
The upgrade should improve structured-output reliability without confusing
PromptKit's bounded deterministic repair with Notarius's existing stage retry
budget or the future feedback-aware semantic-validation loop. PromptKit types
and provider behavior must remain behind Notarius's transport-neutral LLM
boundary.
## Current State
Notarius currently pins PromptKit v0.5.0. Its production adapter prepares one
frozen execution, records credential-redacted details, and runs that same
prepared value. It maps PromptKit capacity failures to an application-owned
error, maps failed structured validation to `ErrInvalidStructuredOutput`, and
returns PromptKit's raw validated bytes and usage metadata.
Every maintained production prompt uses JSON Schema validation and currently
declares `repair_attempts: 0`. Notarius stage bindings separately expose
`retries`, which reruns a complete stage operation after an error or rejected
candidate. The two mechanisms have different ownership and must remain
independent.
Notarius also maintains:
- embedded prompt, schema, and fallback-profile filesystems;
- operator profile-file and profile-directory sources;
- one optional conventional `local` backend registration;
- explicit profile preflight through PromptKit inspection;
- one application-wide scheduled LLM client around the PromptKit adapter;
- PromptKit profile-source fingerprints for checkpoint safety; and
- redacted debug and manifest provenance at application-owned boundaries.
The upgrade must preserve those established responsibilities while revising
the pinned integration contract and any behavior affected by the three
intervening releases.
This roadmap is based on PromptKit's pinned release guides for
[v0.6.0](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.8.0/docs/releases/v0.6.0.md),
[v0.7.0](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.8.0/docs/releases/v0.7.0.md),
and
[v0.8.0](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.8.0/docs/releases/v0.8.0.md),
plus the public API and format documentation at the v0.8.0 tag.
## Target End State
- `go.mod` and `go.sum` pin PromptKit v0.8.0 without a local replacement or
vendored copy.
- Every maintained PromptKit prompt and profile prepares successfully under
v0.8.0's stricter validation and source-loading rules.
- Eligible Notarius structured completions use one PromptKit corrective call by
default after a structurally invalid response. Operators can explicitly set
a value from zero through three for a configured pipeline, with a more local
LLM-backed binding override where needed.
- PromptKit repair remains an inner operation within one Notarius stage
attempt. It never consumes or replenishes the binding's `retries` budget.
- A successful repaired result exposes cumulative usage and the actual repair
count to Notarius's application-owned response and debug models. A repaired
success is not itself a warning.
- Exhausted PromptKit validation remains an invalid structured-output result,
preserving the final candidate and diagnostics for debug and for any
applicable outer Notarius stage policy. Invalid structured output is never
accepted merely because the repair budget was exhausted.
- Profile inheritance, the built-in Rakestrawhome backend/profile, optional
credential behavior, and structured generation errors work through the
existing Notarius PromptKit boundary and are accurately documented.
- Provider-specific PromptKit types do not escape `internal/framework/llm`.
- Checkpoint identity, effective configuration, redacted summaries, and debug
provenance reflect every execution-affecting repair or profile change.
- Current documentation pins and describes v0.8.0; future Notarius semantic
validation retries remain roadmap behavior rather than being conflated with
this dependency upgrade.
## Release-by-Release Adoption
### PromptKit v0.6.0: Correctness, Safety, And Efficiency
PromptKit v0.6.0 adds no public declarations, but intentionally rejects several
formerly permissive or ambiguous inputs. The upgrade must audit Notarius's
embedded and operator-facing integration against these rules:
- YAML `id` and `version` metadata, rather than filenames, define prompt and
profile identity.
- Prompt `content_file` paths are exact, relative, contained paths; built-in
file artifacts must resolve to regular files.
- execution controls, output contracts, and repair budgets must be finite and
within their documented ranges;
- provider endpoints must be absolute HTTP or HTTPS URLs with a host and no
user information, query, or fragment;
- JSON documents and successful provider responses contain exactly one value;
- successful provider responses are bounded to 16 MiB; and
- JSON-compatible values are bounded for depth and expansion.
Notarius should rely on PromptKit for these rules rather than duplicate its
parsers or internal limits. Existing Notarius validation may retain a narrower
application rule where it has independent value, but overlapping validation
must agree with PromptKit and must not accept a value PromptKit will reject
later.
The upgrade automatically receives operation-local schema-plan reuse,
artifact-text memoization, improved cancellation checks, and transport error
identity preservation. Notarius should verify these changes through its real
adapter boundary and avoid adding a second cache or response-body layer that
would duplicate PromptKit's ownership.
### PromptKit v0.7.0: Profiles, Backend Access, And Generation Errors
#### Profile Inheritance
Operator profiles may use `base_profile` to alias or selectively refine a
built-in, fallback, or higher-precedence operator profile. Notarius must pass
profile sources through unchanged and let PromptKit own parent lookup, merge
rules, source precedence, cycle detection, and fully resolved prepared targets.
Preflight inspection must resolve inherited profiles through the same source
and backend composition used at execution. The selected leaf profile ID remains
the public profile identity, while effective backend, endpoint, model, and
reasoning provenance reflect the resolved chain. Notarius must not implement a
second inheritance parser.
The existing complete `dnd-extraction` fallback remains a standalone profile:
PromptKit v0.8.0 does not provide a built-in `openai/gpt-5.6-luna` profile that
would be an appropriate parent. Documentation should nevertheless explain how
operators can use inheritance for environment-specific workload profiles and
should link to PromptKit's pinned format contract rather than duplicate its
field-by-field merge algorithm.
Checkpoint safety must cover inherited behavior. Operator file/directory
digests already cover changes to definitions in those sources, fallback asset
digests cover application parents, and the PromptKit built-in catalog marker
must change from its v0.5.0 identity to v0.8.0 so a changed built-in parent
cannot reuse an incompatible checkpoint.
#### Rakestrawhome Backend And Profile
PromptKit's reserved `rakestrawhome` backend and
`rakestrawhome-gemma-4-31b` profile become available without Notarius-specific
registration. Notarius must not register or shadow the reserved backend ID.
Profile preflight, backend-capacity reporting, scheduling, generation, and
provenance should work for it through the same generic paths used by OpenRouter
and `local`.
The D&D default remains `dnd-extraction`; this upgrade does not silently move a
production workload to Rakestrawhome. Operator documentation should identify
the built-in profile as an available selection and link to PromptKit for its
endpoint, credential environment, model, and capacity defaults.
#### Optional Credentials
An absent or blank optional `APIKeyEnv` now causes PromptKit to omit the
`Authorization` header and send the request. Notarius must not restore the old
failure behavior by pre-reading provider credential environment variables or
by adding provider-specific authentication logic.
Profile inspection may report an explicit `APIKeyRequired` policy without
reading the credential, and execution remains the boundary at which that
requirement is enforced. For optional profiles, an authentication-requiring
provider may instead return a structured 401 or 403 generation failure. The
configuration and operations documentation must explain this distinction.
Notarius does not currently expose PromptKit's in-memory profile-registration
API to operators, and PromptKit's filesystem profile format does not expose
`APIKeyRequired`; therefore Notarius must not promise that an operator profile
can force local credential preflight. Operators should provision the named
environment variable, while Notarius should preserve the provider's structured
authentication failure when it is absent.
Notarius must continue to document mechanisms and environment-variable names,
never secret values.
#### Structured Generation Errors
The adapter should recognize `*promptkit.GenerationError` with `errors.As` and
translate useful information into an immutable, provider-neutral Notarius
error classification. At minimum, retain the HTTP status code so callers and
future retry policy can distinguish transport success with provider rejection
from other generation failures.
PromptKit's provider code, type, and message accessors are bounded but remain
untrusted and potentially sensitive. They must never appear automatically in
ordinary CLI output, warnings, manifests, checkpoint identity, or cache data.
If retained for an explicitly requested debug trace, they must pass through
Notarius's known-secret and bearer redaction and remain clearly identified as
untrusted provider diagnostics. Default error formatting should continue to
use a bounded, redacted application-owned message.
Capacity and cancellation retain their current more specific classifications
and precedence. This upgrade does not add automatic provider-error retry
classification; it only preserves safe structured data needed for diagnosis
and later policy.
### PromptKit v0.8.0: Bounded Structured-Output Repair
#### Default Policy
Every maintained production prompt whose output is consumed as structured data
should declare one repair attempt. All current production prompts use eligible
JSON Schema validation, so no current prompt needs a zero default merely
because of its output mode.
One repair means at most one corrective generation after the initial
candidate. PromptKit reconstructs the immutable original conversation and
appends only the latest invalid assistant candidate and latest deterministic
validation diagnostics. It preserves the selected target, direct session ID,
provider-native structured-output contract, and backend capacity policy. This
shape preserves the original cacheable prompt prefix and avoids accumulating
unbounded failed history.
The default is deliberately small. A single repair captures the common case in
which a capable model can correct malformed JSON or a schema violation after
receiving an exact diagnostic, while bounding the extra latency and cost of a
single structured completion.
#### Configuration Contract
The public configuration is an optional, presence-aware
`structured_output_repair_attempts` integer at pipeline scope and at each
LLM-backed module or validator binding. Its effective precedence is:
1. the binding value, when present;
2. the pipeline value, when present; and
3. the selected prompt's declared `repair_attempts` value.
The value must be from zero through three. Explicit zero disables PromptKit
repair at that scope. A deterministic binding must reject the field because it
cannot perform structured LLM repair. Validator bindings may use it only when
the selected validator is LLM-backed. Shorthand module bindings continue to
inherit the pipeline or prompt default.
The long, provider-neutral name is intentional: it distinguishes PromptKit's
inner structural repair from the existing binding `retries` field, which owns
complete stage attempts, without exposing a dependency name in generic
pipeline contracts.
The effective value must survive file parsing, cloning, redacted summaries,
pipeline resolution, and pipeline digest construction without pointer aliasing
or loss of presence. It must affect checkpoint identity because it can change
the selected result, latency, token usage, and provider cost.
#### Adapter Contract
The transport-neutral structured-completion request should carry an optional
application-owned structural-repair budget. No `promptkit.OutputContract` or
other PromptKit type may cross the adapter boundary.
PromptKit v0.8.0 request validation replaces the complete prompt output
contract rather than merging one field. When Notarius has a configured
override, the adapter must therefore inspect the selected prompt, copy its
normalized declared format, validation mode, and schema path, change only the
repair count, and supply that complete contract on the prepared request. A nil
override continues to use the prompt declaration directly. Inspection and
preparation must use the same immutable engine sources; a small adapter-local
cache keyed by normalized prompt ID and version is acceptable but not required
without measured need.
This approach prevents configuration from accidentally dropping JSON Schema
validation, avoids duplicating schema paths in pipeline YAML, and keeps prompt
assets authoritative for every output-contract field other than the explicit
operator override.
The transport-neutral structured-completion response should report the actual
number of PromptKit repair calls. PromptKit's returned token usage is already
cumulative and must be passed through without re-summing it. Debug records
should distinguish the configured budget from the actual count. Ordinary run
manifests need not gain raw prompt or response data merely to report repairs;
any durable aggregate should be added only if it has a clear consumer contract.
#### Result And Failure Semantics
- A valid initial candidate returns normally with zero actual repairs.
- A valid corrected candidate returns normally with cumulative usage and its
positive actual repair count. It does not emit a warning solely because a
repair occurred.
- Exhausting the repair budget returns PromptKit's final candidate and failed
validation result. The adapter maps this to
`ErrInvalidStructuredOutput`, preserves the response and debug material, and
does not decode or accept the candidate.
- An explicitly empty or whitespace-only candidate participates in the
declared structural validation and repair flow. Missing, `null`, or
non-string provider content remains a malformed provider response.
- A generation failure during a corrective call is an operational generation
failure and uses the same safe structured-error adaptation as an initial
generation failure.
- Context cancellation remains authoritative throughout the initial and
corrective calls.
PromptKit repair happens inside one scheduled `CompleteStructured` operation.
The Notarius scheduler holds one permit for that logical operation while
PromptKit performs its initial and serial corrective calls; PromptKit
reacquires its own selected-backend capacity for each corrective generation.
Because corrective calls are serial, this cannot expand actual concurrent
provider work beyond the number of admitted Notarius operations, but
documentation must stop describing the Notarius permit as a separate admission
event for every internal repair call.
One `CompleteStructured` invocation with effective PromptKit repair budget `R`
may make at most `R + 1` provider calls. If one stage attempt makes `C`
structured-completion invocations, a binding with `retries: N` has an upper
bound of `(N + 1) * C * (R + 1)` provider calls; `C` may itself be a bounded,
data-dependent module property, as it is for batched semantic reconciliation.
LLM-backed validators have their own corresponding invocation counts, budgets,
and costs. These formulas are upper bounds, not promises that every failure is
retryable or that every attempt reaches the provider.
## Profile And Prompt Source Compatibility
The upgrade must preserve Notarius's source precedence: an operator source,
then registered application fallback profiles, then PromptKit built-ins. A
selected malformed definition remains authoritative and fails rather than
falling through. Parent resolution introduced by profile inheritance observes
that same precedence.
All embedded prompt manifests, shared content fragments, response schemas, and
fallback profiles must be prepared or inspected offline under v0.8.0. The
review should specifically catch:
- IDs inferred accidentally from filenames;
- stale or escaping `content_file` paths;
- missing or non-regular embedded artifacts;
- repair values outside zero through three or paired with ineligible
validation;
- schemas or examples that are not exact single JSON documents;
- unsupported endpoint forms; and
- JSON-compatible variables or profile extras that exceed upstream bounds.
No prompt prose, schema shape, durable D&D artifact contract, or default D&D
model should change merely to exercise the dependency. Prompt manifests should
change only as needed to enable the adopted repair default and satisfy v0.8.0
contracts.
## Provenance, Debugging, And Security
- Update the opaque PromptKit built-in profile-catalog identity from v0.5.0 to
v0.8.0. Do not hash or publish PromptKit's internal catalog bytes.
- Ensure a prompt's repair default remains covered by its existing prompt asset
fingerprint and a configured effective override remains covered by the
resolved pipeline digest.
- Preserve selected leaf profile identity while recording the inherited
effective target already exposed by PromptKit inspection and prepared
details.
- Add actual structural-repair count and, when useful, the configured budget to
application-owned debug material. Token totals remain PromptKit's cumulative
values.
- Do not generate a warning for a successful repair. Repair exhaustion is an
invalid-output failure, while provider rejection is a generation failure.
- Never expose raw provider diagnostic fields without explicit debug capture
and application redaction. Do not place them in normal errors or durable
summaries.
- Preserve context and transport error identity sufficiently for
`errors.Is`-based cancellation and deadline handling after adapting the
external error.
## Documentation And Examples
Implementation should update current-state documentation only when the new
behavior lands:
- `docs/integrations/pkg-promptkit.md` must pin v0.8.0 and define the revised
prepared-execution, repair, profile-inheritance, backend, credential, and
error-adaptation boundary.
- `docs/config.md` must own the repair configuration fields, precedence,
allowed range, explicit-zero behavior, profile inheritance availability, and
optional credential semantics.
- `docs/operations.md` must explain structural repair cost, timeout and
concurrency effects, credential failures, and its distinction from stage
retries.
- `docs/internal/llm.md` must describe adapter contract replacement, actual
repair metadata, error adaptation, source compatibility, and scheduling.
- `docs/internal/pipeline.md` must describe how effective repair configuration
is resolved and how inner repair differs from outer stage attempts.
- `docs/policy/architecture.md` should receive only the durable ownership rule:
PromptKit owns bounded deterministic structural repair within one completion,
while Notarius owns stage attempts and semantic validation policy. Detailed
fields and retry formulas belong in their canonical configuration and
operations documents.
Update maintained configuration examples only if the public Notarius
configuration contract changes. A short inheritance illustration may remain in
the configuration reference; do not create a complete example solely to copy
PromptKit's upstream profile catalog. All upstream links must point to the
v0.8.0 tag. Historical release or archived roadmap references should remain
historical.
No ADR is required solely to pin a newer dependency. The durable separation
between PromptKit structural repair and Notarius semantic stage retries should
be stated in architecture documentation now; the more extensive future
validation state machine still warrants the separate ADR already identified in
`future.md` when that work is promoted.
## Validation And Acceptance Criteria
The implementation is complete when:
- the repository builds and tests against PromptKit v0.8.0 with no replacement
directive, workspace dependency, or vendored source;
- every maintained prompt and profile prepares or inspects successfully under
the v0.8.0 source, path, endpoint, output-contract, and JSON-value rules;
- an invalid first JSON Schema candidate followed by a valid correction returns
the valid raw output, cumulative usage, and actual repair count through the
Notarius adapter;
- repair exhaustion returns the final raw candidate and debug material with an
error matching `ErrInvalidStructuredOutput`;
- a corrective generation failure retains safe generation classification and
provider status without leaking untrusted provider detail;
- explicit empty content follows structural validation rather than being
misclassified by Notarius;
- repair configuration is presence-aware, range checked, rejected on
deterministic bindings, resolved with documented precedence, and included in
effective pipeline identity;
- inherited profiles resolve consistently during preflight and execution, and
changes to any relevant operator, fallback, or built-in parent invalidate
checkpoint reuse;
- the Rakestrawhome built-in profile reaches generic preflight, scheduling, and
provenance paths without application-specific registration;
- optional missing credentials and explicitly required credentials behave as
documented without contacting real providers in tests;
- cancellation, timeout, backend capacity, prepared-execution snapshot,
session ID, raw-output, debug-redaction, and existing profile provenance
behavior remain intact;
- maintained examples validate successfully; and
- canonical documentation contains no active v0.5.0 pin or claim that PromptKit
is always single-pass.
Tests should follow `docs/policy/testing.md`: exercise observable Notarius
contracts with offline fake clients or `httptest` boundaries, and do not copy
PromptKit's entire internal repair test suite or assert its exact correction
message prose. The dependency's internal wording is not a Notarius contract.
## Non-Goals
- Implementing Notarius's future feedback-aware semantic stage-retry loop.
- Adding the D&D combat-scene semantic validator.
- Redesigning warning policy or treating successful structural repair as a
warning.
- Adding provider transport retries or deciding which HTTP statuses should
consume a stage retry.
- Exposing PromptKit request, response, profile, validation, capacity, or error
types outside the LLM adapter.
- Changing durable artifact schemas, D&D prompt semantics, the D&D default
model, or the fixed pipeline shape.
- Reimplementing PromptKit profile inheritance, schema validation, response
bounds, repair conversations, backend admission, or provider parsing inside
Notarius.
## Decisions
### 1. Default Structured-Output Repair Budget
**Decision: default to one repair attempt.** Set every maintained
eligible production prompt to `repair_attempts: 1`. One corrective call is a
strong fit for Notarius because every current production LLM response has a
strict JSON Schema contract, smaller cost-effective models are a deliberate
deployment target, and a precise structural diagnostic often makes one retry
materially more successful. The budget is paid only after a structurally
invalid candidate and remains tightly bounded.
**Alternative considered: retain zero by default.** This preserves single-pass
cost and latency and requires operators to opt in. It is preferable for an
environment where every additional request is expensive or where upstream
provider-native schema enforcement already produces negligible invalid output.
It is less suitable as the Notarius default because one malformed response can
otherwise discard substantial completed pipeline work.
**Alternative considered: default to two.** This may improve recovery for
weak models, but it doubles the worst-case corrective cost relative to the
selected default and compounds with outer stage retries. It should be an
operator choice supported by configuration, not the initial default, unless
observational evidence shows that the second correction has a worthwhile
marginal success rate.
### 2. Repair Override Scope
**Decision: support both pipeline and LLM-backed binding overrides.** Use
the presence-aware `structured_output_repair_attempts` field and precedence
defined above. A pipeline value provides the convenient one-line control the
operator requested, while a binding value permits an expensive normalizer or
future LLM-backed validator to use a deliberately different budget. This
mirrors Notarius's established pipeline/binding profile inheritance and scales
without editing embedded prompts.
**Alternative considered: support only a pipeline override.** This is smaller to
implement and document and still permits global enablement or disablement for
one pipeline. Its drawback is that one exceptional prompt cannot opt out or
request a larger budget without changing an embedded asset for every pipeline.
**Alternative considered: expose one global value under the top-level
`promptkit` configuration.** This makes client construction simple, but applies
the same budget to unrelated pipelines and leaks an execution policy into the
dependency configuration block. It is less compositional than pipeline-owned
policy and therefore not recommended.
### 3. Retention Of Provider-Supplied Generation Details
**Decision: retain status in the application-owned error contract and
retain redacted provider code, type, and message only in explicitly requested
debug traces.** Status is useful for diagnosis and future retry policy without
usually containing sensitive data. The other fields can materially explain a
400 response but may echo request or schema content, so they belong only in the
already-sensitive debug surface after Notarius redaction.
**Alternative considered: retain only HTTP status and discard all provider fields.**
This is the safest and smallest policy and still improves typed failure
handling. It sacrifices potentially decisive provider diagnostics, leaving an
operator with less information when a provider returns a terse status and the
problem cannot be reproduced easily.
**Alternative considered: include bounded provider code and type in normal
errors while keeping message debug-only.** Codes and types are often stable and
less sensitive than messages, but PromptKit explicitly classifies every
provider field as untrusted. Promoting them to ordinary output creates a
disclosure and compatibility burden that is not currently justified.

View File

@@ -0,0 +1,282 @@
# Source-Only Releases
## Status
Implemented. Creating the first release under this procedure remains a
separate maintainer operation.
## Purpose
Define a repeatable, guarded release process for Notarius without taking on a
binary-distribution system that its current operator audience does not need.
The process should make an exact source revision, its compatibility impact,
and its validation status easy to identify while keeping installation in the
hands of technically capable operators and deployment automation.
The model is adapted from Weatherreporter's release procedure, but its target
is deliberately narrower: an immutable source tag and checked-in release note
are the release. Notarius does not publish executable archives or support
Windows as part of this work.
## Release Model
Notarius releases come from commits on `main` and use stable semantic-version
tags in the form `vMAJOR.MINOR.PATCH`. Prerelease tags are not part of the
initial process.
Every release has one nonempty, version-matched note at
`docs/releases/<tag>.md`. The note and every affected current-state document
must be present in the tagged commit. The Git tag and checked-in note together
are the durable release record; no separately editable release page is
required.
Published tags are immutable. A maintainer must never move, reuse, or delete a
published tag. If a published candidate is defective, the correction is made
on `main` and released under a new patch version. An unpublished local tag may
be deleted when candidate inspection finds a problem before any remote push.
Before `v1.0.0`, a minor release may intentionally change a documented CLI,
configuration, durable artifact, integration, or operating contract when its
release note explains the impact and required operator action. A patch release
must not intentionally break those documented contracts within its minor
line.
The existing `v0.1.0`, `v0.2.0`, and `v0.3.0` tags remain unchanged. They
predate this procedure and do not need retrospective release notes. The first
release made under this process establishes the release-note series.
## Source-Only Distribution
Notarius does not publish release binaries, archives, installers, container
images, package-manager entries, checksum files, or signatures. A release tag
is suitable for Go-native installation and for an operator-controlled build
from an exact checkout.
The primary installation form is:
```sh
GOWORK=off go install \
gitea.maximumdirect.net/eric/notarius/cmd/notarius@vMAJOR.MINOR.PATCH
```
Operator documentation should also describe cloning the repository, checking
out the tag in detached-head state, and building `./cmd/notarius` with the Go
version declared by `go.mod`. Private-module authentication and `GOPRIVATE`
configuration belong to the operator environment and must be documented by
mechanism rather than with real credentials.
Consumers such as Narratio should pin the desired Notarius tag in provisioning
or deployment configuration. They must continue to decide runtime
compatibility from Notarius's published receipt and artifact schema contracts,
not merely from the executable's product version.
Packaged binaries may be reconsidered if distribution demand, installation
friction, or a broader user audience justifies their build, signing, retention,
and platform-support costs. They are not a prerequisite for a disciplined
release process.
## Platform Policy
Linux is the supported deployment platform. Release validation must run the
test suite and the release build on Linux and must confirm that the command
builds with `CGO_ENABLED=0` for Linux `amd64` and `arm64`.
macOS is a best-effort development and testing platform. Release validation
should confirm that the command cross-compiles with `CGO_ENABLED=0` for Darwin
`amd64` and `arm64`, but the project does not promise packaged artifacts or a
separate runtime test environment for those targets.
Windows is unsupported. The release process must not require Windows builds,
Windows-specific compatibility work, or Windows documentation. Platform-
specific implementation may intentionally use Unix facilities when they are
important to Notarius's filesystem safety and operational model. Any later
decision to support Windows requires its own feature scope and validation
policy.
## Version Reporting
Add a root `notarius --version` interface for deployment diagnostics. It
prints exactly one line:
```text
notarius vMAJOR.MINOR.PATCH
```
when the build has a valid release version, and:
```text
notarius development
```
when no release version is available.
The implementation must obtain the main-module version from Go build
information so `go install ...@vMAJOR.MINOR.PATCH` reports the selected tag. It
must also accept an optional link-time version override so controlled builds
and release CI can identify an exact tag from a checkout. The override must be
validated and must not silently turn arbitrary text into a release version.
Ordinary unversioned checkout builds remain `development`; the release process
must not modify a tracked source constant for each release.
Version reporting is an informational product interface. It does not replace
receipt, configuration, prompt, or artifact schema versioning, and it must not
be used as the sole downstream compatibility check.
## Release Notes
Each new `docs/releases/<tag>.md` document has this minimum structure:
```markdown
# Notarius vMAJOR.MINOR.PATCH
This release ...
## Summary
## Compatibility
## Upgrade
## Changes
```
The note should concisely explain the release's purpose, compatibility with the
preceding release, operator actions, and material user-visible, operational,
integration, and maintainer-visible changes. It should link to canonical
current-state documentation for exact contracts rather than duplicating those
contracts.
Release notes are durable historical summaries. They must not contain
credentials, private infrastructure detail, sensitive campaign material, or
claims that are not true of the tagged candidate. A release note does not
excuse stale current-state documentation; affected canonical documents are
updated in the same candidate.
## Candidate Validation
The release procedure must provide copyable POSIX-shell guards that validate
the release version, release-note filename and heading, required note sections,
repository state, and module hygiene. Validation must be run from the Notarius
repository root with Go workspace behavior disabled.
At minimum, a candidate must pass:
- no tracked `go.work` or `go.work.sum`, no vendored tree, and no `replace`
directive in `go.mod`;
- `GOWORK=off go test -count=1 ./...`;
- `GOWORK=off go test -race -count=1 ./...`;
- `GOWORK=off go vet ./...`;
- `GOWORK=off go build ./...`;
- `GOWORK=off go mod tidy -diff`;
- `gofmt` verification for every tracked Go file;
- `git diff --check` and `git diff --cached --check`;
- validation of both maintained D&D configuration examples with their selected
pipeline;
- Linux `amd64` and `arm64` static command builds;
- best-effort Darwin `amd64` and `arm64` static command builds; and
- a focused manual or automated check that every added or changed local
Markdown link resolves.
The candidate review also checks for generated binaries, test output,
credentials, temporary files, module replacements, vendored dependencies, and
other unintended source-control content. Tests remain offline and do not call
an LLM provider or require live credentials.
## Candidate Publication
The release procedure must guard the exact commit immediately before tagging.
It requires:
- the current branch is `main`;
- the worktree and index are clean;
- the candidate commit has been pushed and exactly matches `origin/main`;
- the matching release note exists in that commit;
- no local or remote tag already uses the selected version; and
- the substantive release checks have passed for that exact candidate.
The maintainer records the exact candidate commit, creates a lightweight tag
bound explicitly to that commit, verifies the local tag target, and pushes only
that tag ref. The procedure must not recommend `git push --tags`.
After publication, the maintainer verifies that the remote tag resolves to the
guarded commit and that the release note can be read from the tagged tree. A
fresh temporary checkout or `go install ...@<tag>` must build successfully, and
the resulting command must report the expected version through `--version`.
## Validation-Only Release Automation
Add a tag-triggered Woodpecker pipeline that validates source releases without
publishing artifacts. It should:
- accept only stable semantic-version tags;
- require the version-matched release note;
- run the same substantive module, test, race, vet, build, formatting, and
whitespace checks as the documented local procedure;
- validate the maintained configuration examples;
- perform the supported and best-effort cross-build checks; and
- verify a release-version build's `notarius --version` output on the CI host.
The pipeline must not upload binaries, create archives or checksums, create or
edit a Gitea release object, or require a release API token. Local guards remain
authoritative before tag publication because CI begins only after the tag is
already remote.
If tag validation fails, preserve the published tag, fix the cause on `main`,
select a new patch version, and repeat the full process. Do not weaken tag
immutability merely because the release contains source rather than binaries.
## Documentation Ownership
In the target state:
- `docs/release.md` owns the maintainer release procedure, commands, ordering,
publication checks, and failure recovery;
- `docs/releases/` owns one historical summary per release made under the new
process;
- `docs/cli.md` owns the `--version` contract;
- `README.md` owns the shortest source-installation example and links to the
release procedure where useful;
- `docs/development.md` routes release preparation, tagging, and verification
work to `docs/release.md`;
- `docs/policy/documentation.md` assigns canonical ownership to the release
procedure and release notes;
- `docs/policy/architecture.md` records Linux support, best-effort macOS
development, unsupported Windows, and source-only distribution only if those
are judged durable development invariants rather than release mechanics; and
- `docs/operations.md` describes only installation or deployment consequences
relevant to operators and links to canonical CLI and release contracts.
Current-state documentation must not describe the new release process,
`--version`, or automated validation until the corresponding behavior exists.
## Acceptance Criteria
- A maintainer can prepare, validate, tag, publish, and verify a source release
by following `docs/release.md` without relying on undocumented knowledge.
- Every new release has an immutable semantic-version tag and matching
checked-in release note in the tagged commit.
- The guarded candidate is clean, synchronized with `origin/main`, and passes
the documented substantive checks before tagging.
- Tag-triggered CI independently validates the published source and never
publishes binary artifacts.
- `go install` of a tagged version succeeds and `notarius --version` reports
that version; ordinary unversioned builds report `development`.
- Linux is the documented supported deployment platform, macOS has a
best-effort development build check, and Windows is explicitly unsupported.
- Downstream compatibility remains based on durable Notarius contracts rather
than the product version alone.
- Existing pre-procedure tags remain untouched and require no invented release
history.
## Non-Goals
- Publishing executable archives, installers, container images, checksums,
signatures, or package-manager entries.
- Supporting or cross-compiling for Windows.
- Creating or maintaining a mutable Gitea release page.
- Supporting prerelease tag syntax in the initial procedure.
- Automating version selection, release-note authorship, commits, or tag
creation.
- Retrospectively creating release notes for `v0.1.0` through `v0.3.0`.
- Treating a product version as a substitute for receipt, configuration,
prompt, or artifact schema compatibility.

2
go.mod
View File

@@ -3,7 +3,7 @@ module gitea.maximumdirect.net/eric/notarius
go 1.25.5
require (
gitea.maximumdirect.net/eric/promptkit v0.5.0
gitea.maximumdirect.net/eric/promptkit v0.8.0
github.com/santhosh-tekuri/jsonschema/v6 v6.0.2
gopkg.in/yaml.v3 v3.0.1
)

4
go.sum
View File

@@ -1,5 +1,5 @@
gitea.maximumdirect.net/eric/promptkit v0.5.0 h1:jnpazLyyNhWrB2xzwwtUkNUfktkTdkENTwuSPnKiYrc=
gitea.maximumdirect.net/eric/promptkit v0.5.0/go.mod h1:R95NM6fbMDGDC0/UomgnSBP6ui2ns+8SZb8bESNvrDQ=
gitea.maximumdirect.net/eric/promptkit v0.8.0 h1:NGd9hDLu0UMxKbvittMrqM5Ua94eFb+kOE7UIir8l08=
gitea.maximumdirect.net/eric/promptkit v0.8.0/go.mod h1:R95NM6fbMDGDC0/UomgnSBP6ui2ns+8SZb8bESNvrDQ=
github.com/dlclark/regexp2 v1.11.0 h1:G/nrcoOa7ZXlpoa/91N3X7mM3r8eIlMBBJZvsz/mxKI=
github.com/dlclark/regexp2 v1.11.0/go.mod h1:DHkYz0B9wPfa6wondMfaivmHpzrQ3v9q8cnmRbL6yW8=
github.com/santhosh-tekuri/jsonschema/v6 v6.0.2 h1:KRzFb2m7YtdldCEkzs6KqmJw4nqEVZGK7IN2kJkjTuQ=

View File

@@ -0,0 +1,36 @@
// Package buildinfo resolves the product version embedded in a Notarius build.
package buildinfo
import (
"fmt"
"regexp"
"runtime/debug"
)
var stableVersion = regexp.MustCompile(`^v(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)$`)
// Override is set at link time for controlled builds.
var Override string
// Version returns the release version embedded in the build, or development
// when the build does not carry a stable release tag.
func Version() (string, error) {
buildVersion := ""
if info, ok := debug.ReadBuildInfo(); ok {
buildVersion = info.Main.Version
}
return resolve(Override, buildVersion)
}
func resolve(override, buildVersion string) (string, error) {
if override != "" {
if !stableVersion.MatchString(override) {
return "", fmt.Errorf("build version override is not a stable release tag")
}
return override, nil
}
if stableVersion.MatchString(buildVersion) {
return buildVersion, nil
}
return "development", nil
}

View File

@@ -0,0 +1,45 @@
package buildinfo
import "testing"
func TestResolve(t *testing.T) {
tests := []struct {
name string
override string
buildVersion string
want string
wantErr bool
}{
{name: "stable main module version", buildVersion: "v1.2.3", want: "v1.2.3"},
{name: "zero version", buildVersion: "v0.0.0", want: "v0.0.0"},
{name: "override takes precedence", override: "v2.3.4", buildVersion: "v1.2.3", want: "v2.3.4"},
{name: "invalid override", override: "version", buildVersion: "v1.2.3", wantErr: true},
{name: "override with whitespace", override: " v1.2.3", wantErr: true},
{name: "leading zero major", buildVersion: "v01.2.3", want: "development"},
{name: "leading zero minor", buildVersion: "v1.02.3", want: "development"},
{name: "leading zero patch", buildVersion: "v1.2.03", want: "development"},
{name: "build version with whitespace", buildVersion: "v1.2.3 ", want: "development"},
{name: "prerelease", buildVersion: "v1.2.3-rc.1", want: "development"},
{name: "build suffix", buildVersion: "v1.2.3+build.1", want: "development"},
{name: "pseudo version", buildVersion: "v0.0.0-20260102030405-abcdef123456", want: "development"},
{name: "development build", buildVersion: "(devel)", want: "development"},
{name: "missing build information", want: "development"},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
got, err := resolve(tt.override, tt.buildVersion)
if tt.wantErr {
if err == nil {
t.Fatal("resolve() error = nil, want error")
}
return
}
if err != nil {
t.Fatalf("resolve() error = %v", err)
}
if got != tt.want {
t.Fatalf("resolve() = %q, want %q", got, tt.want)
}
})
}
}

View File

@@ -0,0 +1,81 @@
package cli
import (
"context"
"errors"
"strings"
"testing"
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
)
const invalidEnemyEventExtractorKey = "test/dnd/invalid-enemy-events"
func TestAssembledEnemyEventLaneRejectsInvalidFinalArtifactDespiteValidatorOverrides(t *testing.T) {
components := productionTestComponents(t)
if err := pipeline.RegisterExtractor[dnd.EnemyEventList](components.registries.Extractors, pipeline.ModuleSpec{
Key: invalidEnemyEventExtractorKey,
Stage: pipeline.StageExtract,
ExecutionClass: contracts.ExecutionClassDeterministic,
Requires: []string{"chunks", "source.transcript"},
Provides: []string{"dnd.enemy_events"},
ArtifactKind: dnd.EnemyEventListKind,
}, func() (contracts.Extractor[dnd.EnemyEventList], error) {
return invalidEnemyEventExtractor{}, nil
}); err != nil {
t.Fatalf("register extractor: %v", err)
}
accept := pipeline.ValidatorOverride{Set: true, Validators: []pipeline.ModuleBinding{pipeline.Binding("generic/always_accept")}}
resolved, err := pipeline.ResolvePipeline(pipeline.PipelineProfile{
ID: "assembled-invalid-enemy-events",
Input: pipeline.Binding("seriatim"),
Chunk: pipeline.ModuleBinding{Module: "generic", Options: map[string]any{"max_units": 1}},
Artifacts: map[string]pipeline.ArtifactLaneProfile{
"enemy-events": {
Extract: pipeline.ModuleBinding{Module: invalidEnemyEventExtractorKey, Validators: accept},
Normalize: pipeline.ModuleBinding{Module: pipeline.DefaultNormalizeModule, Validators: accept},
},
},
Output: pipeline.Binding("json"),
}, pipeline.ResolveOptions{}, catalogFromRegistries(components.registries))
if err != nil {
t.Fatalf("ResolvePipeline() error = %v", err)
}
prepared, err := pipeline.Prepare(resolved, components.registries, pipeline.ModuleDependencies{})
if err != nil {
t.Fatalf("Prepare() error = %v", err)
}
_, err = pipeline.New().Run(context.Background(), pipeline.RunInput{
Prepared: prepared,
RawInput: readRepositoryFile(t, "examples", "seriatim-minimal-transcript.json"),
ChunkCacheMode: pipeline.ChunkCacheBypass,
})
if err == nil || !strings.Contains(err.Error(), "serialize accepted extract output") || !strings.Contains(err.Error(), "must not exceed") {
t.Fatalf("Run() error = %v, want final durable range rejection", err)
}
}
type invalidEnemyEventExtractor struct{}
func (invalidEnemyEventExtractor) Key() string { return invalidEnemyEventExtractorKey }
func (invalidEnemyEventExtractor) ReferenceSlots() []contracts.ReferenceSlot { return nil }
func (invalidEnemyEventExtractor) Extract(ctx context.Context, req contracts.TypedExtractionRequest) (contracts.TypedExtractionResult[dnd.EnemyEventList], error) {
if err := ctx.Err(); err != nil {
return contracts.TypedExtractionResult[dnd.EnemyEventList]{}, err
}
if req.Source == nil {
return contracts.TypedExtractionResult[dnd.EnemyEventList]{}, errors.New("assembled extractor requires source")
}
return contracts.TypedExtractionResult[dnd.EnemyEventList]{Value: dnd.EnemyEventList{Events: []dnd.EnemyEvent{{
Name: "Ashfang",
Kind: dnd.EnemyEventKindEngaged,
SourceRefs: []source.SourceRef{{SourceID: req.Source.ID, StartUnitID: 2, EndUnitID: 1}},
}}}}, nil
}

View File

@@ -134,7 +134,7 @@ func TestAssembledSpellPipelineHonorsNormalizeValidatorOverride(t *testing.T) {
}
}
func TestAssembledSpellPipelineRejectsUnknownSpellWithoutPromotingAttemptWarning(t *testing.T) {
func TestAssembledSpellPipelinePromotesTerminalUnknownSpellWarning(t *testing.T) {
registries, resolved, _ := assembledSpellPipeline(t, assembledSpellPipelineOptions{unknownSpell: true})
prepared, err := pipeline.Prepare(resolved, registries, pipeline.ModuleDependencies{})
if err != nil {
@@ -161,10 +161,8 @@ func TestAssembledSpellPipelineRejectsUnknownSpellWithoutPromotingAttemptWarning
if !reflect.DeepEqual(rejectedFile.Rejected, output.Rejected) {
t.Fatalf("rejected file = %#v, run rejections = %#v, want durable rejection diagnostic", rejectedFile.Rejected, output.Rejected)
}
for _, warning := range output.Warnings {
if warning.ReasonCode == spellnormalize.ReasonCodeSpellNameUnresolved {
t.Fatalf("warnings = %#v, want rejected-attempt warning to remain non-durable", output.Warnings)
}
if len(output.Warnings) != 1 || output.Warnings[0].ReasonCode != spellnormalize.ReasonCodeSpellNameUnresolved || output.Warnings[0].Scope != "spell_casts[0]" {
t.Fatalf("warnings = %#v, want terminal normalize catalog warning", output.Warnings)
}
}

View File

@@ -8,6 +8,8 @@ import (
"path/filepath"
"strings"
"testing"
"gitea.maximumdirect.net/eric/notarius/internal/buildinfo"
)
func TestCommandHelpSpellingsWriteUsageToStdout(t *testing.T) {
@@ -38,6 +40,7 @@ func TestCommandSyntaxErrorsUseStderrAndExitTwo(t *testing.T) {
{name: "unknown pipelines subcommand", args: []string{"pipelines", "unknown"}, want: "unknown pipelines subcommand"},
{name: "malformed run flag", args: []string{"run", "demo", "--chunk_cache", "invalid"}, want: "not supported"},
{name: "unknown flag", args: []string{"config", "validate", "--unknown"}, want: "flag provided but not defined"},
{name: "version arguments", args: []string{"--version", "extra"}, want: "--version does not accept arguments"},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
@@ -50,6 +53,33 @@ func TestCommandSyntaxErrorsUseStderrAndExitTwo(t *testing.T) {
}
}
func TestCommandVersionOutput(t *testing.T) {
previous := buildinfo.Override
t.Cleanup(func() { buildinfo.Override = previous })
tests := []struct {
name string
override string
wantCode int
wantStdout string
wantStderr string
}{
{name: "development", wantStdout: "notarius development\n"},
{name: "release override", override: "v1.2.3", wantStdout: "notarius v1.2.3\n"},
{name: "invalid override", override: "release", wantCode: 1, wantStderr: "notarius: build version override is not a stable release tag\n"},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
buildinfo.Override = tt.override
var stdout, stderr bytes.Buffer
code := RunWithOptions([]string{"--version"}, &stdout, &stderr, Options{})
if code != tt.wantCode || stdout.String() != tt.wantStdout || stderr.String() != tt.wantStderr {
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
}
})
}
}
func TestConfigDiscoveryPrefersExplicitPathThenEnvironment(t *testing.T) {
explicit := writeCommandConfig(t, "explicit", "alpha")
environment := writeCommandConfig(t, "environment", "beta")

View File

@@ -189,10 +189,18 @@ func TestMaintainedCompleteExamplePublishesRegistryBackedEntityOccurrences(t *te
}
evidence := readProductionJSON[evidencecontext.Document](t, filepath.Join(runRoot, "evidence-context.json"))
for _, laneID := range []string{"enemy-events", "npc-registry", "npc-occurrences", "item-registry", "item-occurrences", "location-registry", "location-occurrences"} {
if !containsString(evidence.SelectedLanes, laneID) || !evidenceHasLane(evidence, laneID) {
t.Fatalf("evidence context = %#v, want direct %s evidence", evidence, laneID)
if len(evidence) == 0 {
t.Fatalf("evidence context = %#v, want selected source-unit evidence", evidence)
}
seenEvidenceUnits := make(map[int]struct{}, len(evidence))
for _, unit := range evidence {
if unit.Ref.SourceID != "session-ravenfall" || unit.Ref.StartUnitID != unit.ID || unit.Ref.EndUnitID != unit.ID {
t.Fatalf("evidence unit = %#v, want unchanged source-unit self-reference", unit)
}
if _, exists := seenEvidenceUnits[unit.ID]; exists {
t.Fatalf("evidence context = %#v, want each source unit once", evidence)
}
seenEvidenceUnits[unit.ID] = struct{}{}
}
requests := client.requestsFor(enemyevents.PromptID)
@@ -219,14 +227,15 @@ func TestMaintainedCompleteExamplePublishesRegistryBackedEntityOccurrences(t *te
}
for _, request := range locationRequests {
registryInput := request.Inputs["location_registry"]
if !strings.Contains(string(registryInput.Content), "Moon Gate") || !strings.Contains(string(registryInput.Content), `"id"`) || strings.Contains(string(registryInput.Content), "source_refs") {
t.Fatalf("location occurrence registry input = %q, want source-free ID grounding", registryInput.Content)
if !strings.Contains(string(registryInput.Content), "Moon Gate") || !strings.Contains(string(registryInput.Content), "registry_refs") || strings.Contains(string(registryInput.Content), `"id"`) || strings.Contains(string(registryInput.Content), "source_refs") {
t.Fatalf("location occurrence registry input = %q, want contextual selector grounding", registryInput.Content)
}
}
for _, test := range []struct {
promptID string
slot string
name string
promptID string
slot string
name string
requiresIDs bool
}{
{promptID: npcoccurrences.PromptID, slot: "npc_registry", name: "Kesh"},
{promptID: itemoccurrences.PromptID, slot: "item_registry", name: "Moonblade"},
@@ -237,8 +246,9 @@ func TestMaintainedCompleteExamplePublishesRegistryBackedEntityOccurrences(t *te
}
for _, request := range requests {
registryInput := request.Inputs[test.slot]
if !strings.Contains(string(registryInput.Content), test.name) || !strings.Contains(string(registryInput.Content), `"id"`) || strings.Contains(string(registryInput.Content), "source_refs") {
t.Fatalf("%s registry input = %q, want source-free ID grounding", test.promptID, registryInput.Content)
hasID := strings.Contains(string(registryInput.Content), `"id"`)
if !strings.Contains(string(registryInput.Content), test.name) || hasID != test.requiresIDs || strings.Contains(string(registryInput.Content), "source_refs") {
t.Fatalf("%s registry input = %q, want source-free configured grounding", test.promptID, registryInput.Content)
}
}
}
@@ -319,16 +329,16 @@ func (client *enemyEventLLMClient) CompleteStructured(ctx context.Context, reque
} else {
var registry struct {
Items []struct {
ID string `json:"id"`
Name string `json:"name"`
} `json:"items"`
}
if err := json.Unmarshal(request.Inputs["item_registry"].Content, &registry); err != nil {
return contracts.StructuredCompletionResponse{}, fmt.Errorf("decode generated item registry: %w", err)
}
if len(registry.Items) != 1 {
if len(registry.Items) != 1 || registry.Items[0].Name != "Moonblade" {
return contracts.StructuredCompletionResponse{}, fmt.Errorf("generated item registry has %d items, want 1", len(registry.Items))
}
content = []byte(fmt.Sprintf(`{"occurrences":[{"item_id":%q,"name":"Moonblade","kind":"discovered","quantity":null,"from":null,"to":null,"source_refs":[{"start_segment":5,"end_segment":5}]}]}`, registry.Items[0].ID))
content = []byte(`{"occurrences":[{"name":"Moonblade","kind":"discovered","quantity":null,"from":null,"to":null,"source_refs":[{"start_unit_id":5,"end_unit_id":5}]}]}`)
}
case combat.PromptID:
content = []byte(`{"combat_turns":[{"actor":"Kesh","turn_kind":"turn","source_refs":[{"start_unit_id":8,"end_unit_id":8}]}]}`)
@@ -336,23 +346,27 @@ func (client *enemyEventLLMClient) CompleteStructured(ctx context.Context, reque
if combatScene {
var registry struct {
NPCs []struct {
ID string `json:"id"`
Name string `json:"name"`
} `json:"npcs"`
}
if err := json.Unmarshal(request.Inputs["npc_registry"].Content, &registry); err != nil {
return contracts.StructuredCompletionResponse{}, fmt.Errorf("decode generated NPC registry: %w", err)
}
if len(registry.NPCs) == 0 {
if len(registry.NPCs) == 0 || registry.NPCs[0].Name != "Kesh" {
return contracts.StructuredCompletionResponse{}, fmt.Errorf("generated NPC registry has no NPCs")
}
content = []byte(fmt.Sprintf(`{"occurrences":[{"npc_id":%q,"name":"Kesh","kind":"combat_opponent","source_refs":[{"start_unit_id":7,"end_unit_id":7}]}]}`, registry.NPCs[0].ID))
content = []byte(`{"occurrences":[{"name":"Kesh","kind":"combat_opponent","source_refs":[{"start_unit_id":7,"end_unit_id":7}]}]}`)
} else {
content = []byte(`{"occurrences":[]}`)
}
case locationoccurrences.PromptID:
var registry struct {
Locations []struct {
ID string `json:"id"`
Name string `json:"name"`
RegistryRefs []struct {
StartUnitID int `json:"start_unit_id"`
EndUnitID int `json:"end_unit_id"`
} `json:"registry_refs"`
} `json:"locations"`
}
if err := json.Unmarshal(request.Inputs["location_registry"].Content, &registry); err != nil {
@@ -362,14 +376,18 @@ func (client *enemyEventLLMClient) CompleteStructured(ctx context.Context, reque
return contracts.StructuredCompletionResponse{}, fmt.Errorf("generated location registry has no locations")
}
unitID := 1
locationID := registry.Locations[0].ID
location := registry.Locations[0]
if combatScene {
unitID = 7
if len(registry.Locations) > 1 {
locationID = registry.Locations[1].ID
location = registry.Locations[1]
}
}
content = []byte(fmt.Sprintf(`{"occurrences":[{"location_id":%q,"name":"Moon Gate","kind":"visited","source_refs":[{"start_unit_id":%d,"end_unit_id":%d}]}]}`, locationID, unitID, unitID))
registryRefs, err := json.Marshal(location.RegistryRefs)
if err != nil {
return contracts.StructuredCompletionResponse{}, fmt.Errorf("encode location selector: %w", err)
}
content = []byte(fmt.Sprintf(`{"occurrences":[{"name":%q,"registry_refs":%s,"kind":"visited","source_refs":[{"start_unit_id":%d,"end_unit_id":%d}]}]}`, location.Name, registryRefs, unitID, unitID))
case enemyevents.PromptID:
content = []byte(`{"events":[{"name":"Kesh","kind":"fled","source_refs":[{"start_unit_id":10,"end_unit_id":10}]}]}`)
default:
@@ -405,17 +423,6 @@ func containsString(values []string, want string) bool {
return false
}
func evidenceHasLane(value evidencecontext.Document, laneID string) bool {
for _, context := range value.Contexts {
for _, reference := range context.EvidenceRefs {
if reference.LaneID == laneID {
return true
}
}
}
return false
}
func generatedReferenceBinding(bindings []pipeline.ReferenceBinding, slotName string) (pipeline.ReferenceBinding, bool) {
for _, binding := range bindings {
if binding.SlotName == slotName && binding.Artifact != nil {

View File

@@ -19,12 +19,15 @@ import (
"testing/fstest"
"time"
"gopkg.in/yaml.v3"
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
"gitea.maximumdirect.net/eric/notarius/internal/framework/chunkmap"
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/chunk/scenes"
combatcodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/combatturns"
@@ -311,6 +314,73 @@ func TestProductionPromptAssetsPrepareWithoutProviderCredentials(t *testing.T) {
if _, err := pipeline.Prepare(effective.ResolvedPipeline, components.registries, pipeline.ModuleDependencies{LLM: &productionFakeLLMClient{}}); err != nil {
t.Fatalf("prepare production scene and spell modules: %v", err)
}
schemaFS, err := components.assets.SchemaFS()
if err != nil {
t.Fatalf("production schema assets: %v", err)
}
if _, err := fs.ReadFile(schemaFS, filepath.Base(semanticreconcile.SchemaAssetPath)); err != nil {
t.Fatalf("generic reconciliation schema asset: %v", err)
}
options, err := components.assets.PromptKitOptions()
if err != nil {
t.Fatalf("production PromptKit options: %v", err)
}
options = append(options, promptkit.WithProfiles(promptkit.OpenAICompatibleProfile(promptkit.OpenAICompatibleProfileConfig{
ID: "assembled-prompt-test", Endpoint: "http://127.0.0.1:1/v1", Model: "test",
})))
engine, err := promptkit.NewEngine(promptkit.Config{}, options...)
if err != nil {
t.Fatalf("production prompt engine: %v", err)
}
promptFS, err := components.assets.PromptFS()
if err != nil {
t.Fatalf("production prompt assets: %v", err)
}
type manifest struct {
ID string `yaml:"id"`
Version string `yaml:"version"`
Inputs []struct {
Name string `yaml:"name"`
} `yaml:"inputs"`
}
preparedPrompts := 0
if err := fs.WalkDir(promptFS, ".", func(path string, entry fs.DirEntry, walkErr error) error {
if walkErr != nil {
return walkErr
}
if entry.IsDir() || filepath.Base(path) != "prompt.yaml" {
return nil
}
data, err := fs.ReadFile(promptFS, path)
if err != nil {
return err
}
var prompt manifest
if err := yaml.Unmarshal(data, &prompt); err != nil {
return err
}
inputs := make(map[string]promptkit.ArtifactRef, len(prompt.Inputs))
for _, input := range prompt.Inputs {
inputs[input.Name] = promptkit.Inline(`{}`)
}
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{
PromptID: prompt.ID, PromptVersion: prompt.Version, ProfileID: "assembled-prompt-test", Inputs: inputs,
})
if err != nil {
return fmt.Errorf("prepare production prompt %q: %w", prompt.ID, err)
}
if prepared.OutputContract.RepairAttempts != 1 {
return fmt.Errorf("prompt %q repair attempts = %d, want 1", prompt.ID, prepared.OutputContract.RepairAttempts)
}
preparedPrompts++
return nil
}); err != nil {
t.Fatal(err)
}
if preparedPrompts == 0 {
t.Fatal("prepared no production prompts")
}
}
func TestProductionSpellValidatorsPrepareFromMaterializedCatalog(t *testing.T) {

View File

@@ -71,10 +71,10 @@ api_key_env: NOTARIUS_PROMPTKIT_PROFILE_INSPECTION_TEST_KEY
},
{
name: "malformed profile",
profilePath: writeProfile(t, "malformed-profile", "id: malformed-profile\nbackend: [\n"),
profilePath: writeProfile(t, "malformed-profile", "id: malformed-profile\nendpoint: https://provider.example/v1?credential=forbidden\nmodel: malformed-model\n"),
profileID: "malformed-profile",
wantErr: []string{`PromptKit profile "malformed-profile" is invalid or unreadable`},
rejectErr: []string{"malformed-profile.yaml", "backend: ["},
rejectErr: []string{"malformed-profile.yaml", "credential=forbidden"},
},
{
name: "invalid profile source",
@@ -157,3 +157,25 @@ func TestExplicitPromptKitProfileValidationUsesFallbackAssets(t *testing.T) {
t.Fatalf("validateExplicitPromptKitProfiles() error = %v, want nil", err)
}
}
func TestExplicitPromptKitProfileValidationRejectsInvalidInheritanceBeforeGeneration(t *testing.T) {
for _, profiles := range []string{
"id: child\nbase_profile: missing\n",
"id: first\nbase_profile: second\n\n---\nid: second\nbase_profile: first\n",
} {
t.Run("invalid inheritance", func(t *testing.T) {
profilePath := filepath.Join(t.TempDir(), "profiles.yaml")
if err := os.WriteFile(profilePath, []byte(profiles), 0o600); err != nil {
t.Fatal(err)
}
profileID := "child"
if strings.Contains(profiles, "id: first") {
profileID = "first"
}
err := validateExplicitPromptKitProfiles(context.Background(), config.Config{PromptKit: config.PromptKitConfig{ProfileFile: profilePath}}, []string{profileID}, nil)
if err == nil || !strings.Contains(err.Error(), "invalid or unreadable") || strings.Contains(err.Error(), profilePath) {
t.Fatalf("profile preflight error = %v", err)
}
})
}
}

View File

@@ -400,6 +400,9 @@ func (referenceContractCodecA) Encode(stateTestArtifact) ([]byte, error) {
func (referenceContractCodecA) Decode([]byte) (stateTestArtifact, error) {
return stateTestArtifact{Value: "ok"}, nil
}
func (codec referenceContractCodecA) DecodeCandidate(content []byte) (stateTestArtifact, error) {
return codec.Decode(content)
}
func (referenceContractCodecB) Kind() contracts.ArtifactKind { return referenceContractKindBeta }
func (referenceContractCodecB) Schema() contracts.ArtifactSchema {
@@ -415,6 +418,9 @@ func (referenceContractCodecB) Encode(stateTestArtifact) ([]byte, error) {
func (referenceContractCodecB) Decode([]byte) (stateTestArtifact, error) {
return stateTestArtifact{Value: "ok"}, nil
}
func (codec referenceContractCodecB) DecodeCandidate(content []byte) (stateTestArtifact, error) {
return codec.Decode(content)
}
func referenceContractLane(t *testing.T, resolved pipeline.ResolvedPipeline, id string) pipeline.ResolvedArtifactLane {
t.Helper()

View File

@@ -16,9 +16,11 @@ import (
"strings"
"time"
"gitea.maximumdirect.net/eric/notarius/internal/buildinfo"
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
"gitea.maximumdirect.net/eric/notarius/internal/core/debugbundle"
"gitea.maximumdirect.net/eric/notarius/internal/core/fileio"
"gitea.maximumdirect.net/eric/notarius/internal/framework/checkpoint"
"gitea.maximumdirect.net/eric/notarius/internal/framework/chunkplan"
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
@@ -30,6 +32,7 @@ import (
const defaultConfigPath = "/usr/local/etc/notarius/config.yml"
const usage = `Usage:
notarius help
notarius --version
notarius run <pipeline-id> --input path/to/source.json [--json] [flags]
notarius config validate --config path/to/config.yml [--pipeline pipeline-id] [--only lane-a,lane-b]
notarius pipelines list --config path/to/config.yml [--json]
@@ -61,6 +64,20 @@ func Run(args []string, stdout, stderr io.Writer) int {
}
func RunWithOptions(args []string, stdout, stderr io.Writer, opts Options) int {
if len(args) > 0 && args[0] == "--version" {
if len(args) != 1 {
fmt.Fprintln(stderr, "notarius: --version does not accept arguments")
return 2
}
version, err := buildinfo.Version()
if err != nil {
fmt.Fprintf(stderr, "notarius: %v\n", err)
return 1
}
fmt.Fprintf(stdout, "notarius %s\n", version)
return 0
}
var err error
opts, err = normalizeOptions(opts)
if err != nil {
@@ -147,7 +164,7 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
machineOutput := fs.Bool("json", false, "write the successful run result as JSON")
debug := fs.Bool("debug", false, "write a debug bundle")
debugDir := fs.String("debug-dir", "", "debug bundle directory")
llmProfile := fs.String("llm-profile", "", "LLM profile override")
llmProfile := singleValueFlag{name: "--llm-profile"}
reasoningEffort := singleValueFlag{name: "--reasoning-effort"}
clearReasoningEffort := fs.Bool("clear-reasoning-effort", false, "clear the LLM profile reasoning effort")
resume := fs.Bool("resume", false, "reuse compatible recorded checkpoints")
@@ -157,6 +174,7 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
referenceFlags := stringListFlag{}
withoutReferenceFlags := stringListFlag{}
fs.Var(&requestedSessionID, "session-id", "prompt session identifier")
fs.Var(&llmProfile, "llm-profile", "LLM profile override")
fs.Var(&reasoningEffort, "reasoning-effort", "reasoning effort override")
fs.Var(&chunkCache, "chunk_cache", "chunk plan cache mode: auto, bypass, or refresh")
fs.Var(&referenceFlags, "reference", "reference binding, as slot=path, chunk.slot=path, merge.slot=path, lane.slot=path, lane.extract.slot=path, lane.merge.slot=path, or lane.normalize.slot=path")
@@ -203,6 +221,10 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
fmt.Fprintln(stderr, "notarius: --session-id must not be empty")
return 2
}
if llmProfile.set && strings.TrimSpace(llmProfile.value) == "" {
fmt.Fprintln(stderr, "notarius: --llm-profile must not be empty")
return 2
}
if reasoningEffort.set && *clearReasoningEffort {
fmt.Fprintln(stderr, "notarius: --reasoning-effort cannot be combined with --clear-reasoning-effort")
return 2
@@ -338,7 +360,7 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
PipelineID: pipelineID,
Only: only,
Catalog: catalog,
LLMProfileOverride: *llmProfile,
LLMProfileOverride: strings.TrimSpace(llmProfile.value),
ReferenceOverrides: referenceOverrides,
ReferenceUnbinds: referenceUnbinds,
})
@@ -426,7 +448,7 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
if err != nil {
return failPipelineCommand(stderr, commandState, terminalWriter, err)
}
checkpointRecorder, checkpointLoader, err := checkpointHandlersForRun(effective.Config.Cache.Checkpoints, opts, effective.ResolvedPipeline, prepared.CheckpointFingerprints(), llmFingerprints, rawInput, only, llmProfiles, strings.TrimSpace(*llmProfile), effectiveSessionID, runtimeOverrides, *resume)
checkpointRecorder, checkpointLoader, err := checkpointHandlersForRun(effective.Config.Cache.Checkpoints, opts, effective.ResolvedPipeline, prepared.CheckpointFingerprints(), llmFingerprints, rawInput, only, llmProfiles, strings.TrimSpace(llmProfile.value), effectiveSessionID, runtimeOverrides, *resume)
if err != nil {
return failPipelineCommand(stderr, commandState, terminalWriter, err)
}
@@ -752,17 +774,10 @@ func configSource(configPath string) string {
}
func writeOutputFiles(runOutputDir string, files []contracts.OutputFile) error {
type outputTarget struct {
path string
file contracts.OutputFile
}
targets := make([]outputTarget, 0, len(files))
for _, file := range files {
targetPath, err := outputFilePath(runOutputDir, file.Name)
if err != nil {
if _, err := outputFilePath(runOutputDir, file.Name); err != nil {
return err
}
targets = append(targets, outputTarget{path: targetPath, file: file})
}
outputParent := filepath.Dir(runOutputDir)
@@ -775,12 +790,9 @@ func writeOutputFiles(runOutputDir string, files []contracts.OutputFile) error {
}
return fmt.Errorf("create output run directory %q: %w", runOutputDir, err)
}
for _, target := range targets {
if err := os.MkdirAll(filepath.Dir(target.path), 0o755); err != nil {
return fmt.Errorf("create output directory %q: %w", filepath.Dir(target.path), err)
}
if err := writeFileAtomic(target.path, target.file.Bytes, 0o644); err != nil {
return fmt.Errorf("write output file %q: %w", target.file.Name, err)
for _, file := range files {
if err := fileio.WriteBytes(runOutputDir, file.Name, file.Bytes, 0o755, 0o644); err != nil {
return fmt.Errorf("write output file %q: %w", file.Name, err)
}
}
return nil
@@ -823,38 +835,6 @@ func outputFilePath(runOutputDir, logicalName string) (string, error) {
return target, nil
}
func writeFileAtomic(path string, data []byte, perm os.FileMode) error {
dir := filepath.Dir(path)
temp, err := os.CreateTemp(dir, "."+filepath.Base(path)+".tmp-*")
if err != nil {
return err
}
tempPath := temp.Name()
removeTemp := true
defer func() {
if removeTemp {
_ = os.Remove(tempPath)
}
}()
if _, err := temp.Write(data); err != nil {
_ = temp.Close()
return err
}
if err := temp.Chmod(perm); err != nil {
_ = temp.Close()
return err
}
if err := temp.Close(); err != nil {
return err
}
if err := os.Rename(tempPath, path); err != nil {
return err
}
removeTemp = false
return nil
}
func reorderRunArgs(args []string) []string {
var flags []string
var positionals []string

View File

@@ -43,6 +43,12 @@ func TestRunControlsRejectSyntaxWithoutAllocatingState(t *testing.T) {
{name: "blank session ID", args: func(roots stateTestRoots) []string {
return []string{"run", "sample", "--config", roots.config, "--input", roots.input, "--session-id", ""}
}},
{name: "blank LLM profile", args: func(roots stateTestRoots) []string {
return []string{"run", "sample", "--config", roots.config, "--input", roots.input, "--llm-profile", ""}
}},
{name: "whitespace LLM profile", args: func(roots stateTestRoots) []string {
return []string{"run", "sample", "--config", roots.config, "--input", roots.input, "--llm-profile", " \t "}
}},
{name: "multiple pipeline IDs", args: func(roots stateTestRoots) []string {
return []string{"run", "sample", "extra", "--config", roots.config, "--input", roots.input}
}},
@@ -253,7 +259,7 @@ func TestRunLLMProfileOverrideAndValidationUseInjectedBoundaries(t *testing.T) {
return nil, nil, nil
}
var stdout, stderr bytes.Buffer
code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass", "--llm-profile", "override-profile"}, &stdout, &stderr, opts)
code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass", "--llm-profile", " override-profile "}, &stdout, &stderr, opts)
if code != 0 || stderr.Len() != 0 {
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
}

View File

@@ -38,10 +38,36 @@ func TestWriteOutputFilesSupportsNestedLogicalPaths(t *testing.T) {
if err := writeOutputFiles(runPath, []contracts.OutputFile{{Name: "nested/result.json", Bytes: []byte("result")}}); err != nil {
t.Fatal(err)
}
data, err := os.ReadFile(filepath.Join(runPath, "nested", "result.json"))
resultPath := filepath.Join(runPath, "nested", "result.json")
data, err := os.ReadFile(resultPath)
if err != nil || string(data) != "result" {
t.Fatalf("nested output = %q, %v", data, err)
}
for path, want := range map[string]os.FileMode{runPath: 0o755, filepath.Join(runPath, "nested"): 0o755, resultPath: 0o644} {
info, err := os.Stat(path)
if err != nil {
t.Fatal(err)
}
mode := info.Mode().Perm()
if mode&^want != 0 {
t.Fatalf("%s mode = %#o, must not be broader than %#o", path, mode, want)
}
if info.IsDir() && mode&0o700 != 0o700 {
t.Fatalf("%s mode = %#o, want owner access", path, mode)
}
if !info.IsDir() && mode != want {
t.Fatalf("%s mode = %#o, want %#o", path, mode, want)
}
}
entries, err := os.ReadDir(filepath.Join(runPath, "nested"))
if err != nil {
t.Fatal(err)
}
for _, entry := range entries {
if strings.Contains(entry.Name(), ".tmp-") {
t.Fatalf("temporary file remains: %s", entry.Name())
}
}
}
func TestWriteOutputFilesRejectsUnsafeNamesBeforeAllocatingRunDirectory(t *testing.T) {
@@ -75,7 +101,7 @@ func TestWriteOutputFilesRetainsNewPartialDirectoryAndPreservesSibling(t *testin
{Name: "blocked", Bytes: []byte("partial output")},
{Name: "blocked/nested.json", Bytes: []byte("unreachable")},
})
if err == nil || !strings.Contains(err.Error(), "create output directory") {
if err == nil || !strings.Contains(err.Error(), `write output file "blocked/nested.json"`) {
t.Fatalf("writeOutputFiles() error = %v, want later directory failure", err)
}
if got, err := os.ReadFile(filepath.Join(runPath, "blocked")); err != nil || string(got) != "partial output" {

View File

@@ -410,8 +410,8 @@ func TestMaintainedProductionOverlayRunAlignsGroundingValidationAndProvenance(t
t.Fatalf("spell requests = %d, want one", len(requests))
}
catalogInput, ok := requests[0].Inputs[spellcatalog.SpellCatalogReferenceSlot]
if !ok || !strings.Contains(string(catalogInput.Content), "Aegis of Emberfall") || strings.Contains(string(catalogInput.Content), "Emberfall Aegis") {
t.Fatalf("spell catalog prompt input = %#v, want canonical overlay name without alias", catalogInput)
if !ok || !strings.Contains(string(catalogInput.Content), `"canonical_name":"Aegis of Emberfall"`) || !strings.Contains(string(catalogInput.Content), `"aliases":["Emberfall Aegis"]`) {
t.Fatalf("spell catalog prompt input = %#v, want canonical overlay name and recognition alias", catalogInput)
}
artifact := readProductionJSON[dnd.SpellList](t, filepath.Join(runRoot, "lanes", "spells.json"))
if len(artifact.SpellCasts) != 1 || artifact.SpellCasts[0].Spell != "Aegis of Emberfall" {

View File

@@ -92,7 +92,7 @@ func TestProductionSpellCatalogValidationRetries(t *testing.T) {
t.Fatalf("rejection = %#v, want exhausted unknown-spell rejection", rejection)
}
if len(output.Warnings) != 0 {
t.Fatalf("warnings = %#v, want no warnings from rejected attempts", output.Warnings)
t.Fatalf("warnings = %#v, want no emitted warnings from rejected attempts", output.Warnings)
}
return
}

View File

@@ -955,6 +955,9 @@ func (stateTestCodec) Encode(v stateTestArtifact) ([]byte, error) {
func (stateTestCodec) Decode([]byte) (stateTestArtifact, error) {
return stateTestArtifact{Value: "ok"}, nil
}
func (codec stateTestCodec) DecodeCandidate(content []byte) (stateTestArtifact, error) {
return codec.Decode(content)
}
type stateTestExtractor struct{ harness *stateTestHarness }

View File

@@ -116,6 +116,10 @@ func (c *ConcurrencyConfig) recomputeStageWorkerDefaults() {
func clonePipelineProfile(in pipeline.PipelineProfile) pipeline.PipelineProfile {
out := in
if in.StructuredOutputRepairAttempts != nil {
value := *in.StructuredOutputRepairAttempts
out.StructuredOutputRepairAttempts = &value
}
out.Input = cloneModuleBinding(in.Input)
out.Chunk = cloneModuleBinding(in.Chunk)
out.Output = cloneModuleBinding(in.Output)
@@ -196,6 +200,10 @@ func cloneReferenceSource(in pipeline.ReferenceSource) pipeline.ReferenceSource
func cloneModuleBinding(in pipeline.ModuleBinding) pipeline.ModuleBinding {
out := in
if in.StructuredOutputRepairAttempts != nil {
value := *in.StructuredOutputRepairAttempts
out.StructuredOutputRepairAttempts = &value
}
if len(in.Options) > 0 {
out.Options = cloneOptions(in.Options)
}

View File

@@ -427,6 +427,10 @@ func (effectiveCodec) Decode(content []byte) (effectiveArtifact, error) {
return value, err
}
func (codec effectiveCodec) DecodeCandidate(content []byte) (effectiveArtifact, error) {
return codec.Decode(content)
}
type effectiveInput struct{ key string }
func (m effectiveInput) Key() string { return m.key }

View File

@@ -3,6 +3,7 @@ package config
import (
"bytes"
"fmt"
"io"
"os"
"path/filepath"
"sort"
@@ -34,23 +35,27 @@ type FilePromptKitLocalBackendConfig struct {
}
type FilePipelineProfile struct {
LLMProfile *string `yaml:"llm_profile,omitempty"`
Input fileModuleBinding `yaml:"input"`
Chunk *fileModuleBinding `yaml:"chunk,omitempty"`
Artifacts map[string]FileArtifactLaneProfile `yaml:"artifacts,omitempty"`
Steps []FilePipelineStepProfile `yaml:"steps,omitempty"`
Output *fileModuleBinding `yaml:"output,omitempty"`
References map[string]fileReferenceSource `yaml:"references,omitempty"`
artifactsSet bool `yaml:"-"`
stepsSet bool `yaml:"-"`
llmProfileSet bool `yaml:"-"`
LLMProfile *string `yaml:"llm_profile,omitempty"`
StructuredOutputRepairAttempts *int `yaml:"structured_output_repair_attempts,omitempty"`
Input fileModuleBinding `yaml:"input"`
Chunk *fileModuleBinding `yaml:"chunk,omitempty"`
Artifacts map[string]FileArtifactLaneProfile `yaml:"artifacts,omitempty"`
Steps []FilePipelineStepProfile `yaml:"steps,omitempty"`
Output *fileModuleBinding `yaml:"output,omitempty"`
References map[string]fileReferenceSource `yaml:"references,omitempty"`
artifactsSet bool `yaml:"-"`
stepsSet bool `yaml:"-"`
llmProfileSet bool `yaml:"-"`
}
func (p *FilePipelineProfile) UnmarshalYAML(node *yaml.Node) error {
if err := validateStructuredOutputRepairAttemptsNode(node, "pipeline profile"); err != nil {
return err
}
type plainFilePipelineProfile FilePipelineProfile
var decoded plainFilePipelineProfile
seen, err := decodeKnownMapping(node, &decoded, map[string]struct{}{
"llm_profile": {}, "input": {}, "chunk": {}, "artifacts": {}, "steps": {}, "output": {}, "references": {},
"llm_profile": {}, "structured_output_repair_attempts": {}, "input": {}, "chunk": {}, "artifacts": {}, "steps": {}, "output": {}, "references": {},
}, "pipeline profile")
if err != nil {
return err
@@ -146,12 +151,13 @@ type FileDebugConfig struct {
}
type fileModuleBinding struct {
Module string
LLMProfile string
Retries int
Options map[string]any
References map[string]fileReferenceSource
Validators pipeline.ValidatorOverride
Module string
LLMProfile string
StructuredOutputRepairAttempts *int
Retries int
Options map[string]any
References map[string]fileReferenceSource
Validators pipeline.ValidatorOverride
}
type fileReferenceSource struct {
@@ -263,6 +269,12 @@ func (b *fileModuleBinding) UnmarshalYAML(node *yaml.Node) error {
if b.LLMProfile == "" {
return fmt.Errorf("llm_profile must not be empty when set")
}
case "structured_output_repair_attempts":
attempts, err := parseStructuredOutputRepairAttempts(valueNode, "module binding")
if err != nil {
return err
}
b.StructuredOutputRepairAttempts = attempts
case "retries":
var retries int
if err := valueNode.Decode(&retries); err != nil {
@@ -303,15 +315,56 @@ func (b *fileModuleBinding) UnmarshalYAML(node *yaml.Node) error {
func (b fileModuleBinding) toPipelineBinding() pipeline.ModuleBinding {
return pipeline.ModuleBinding{
Module: strings.TrimSpace(b.Module),
LLMProfile: strings.TrimSpace(b.LLMProfile),
Retries: b.Retries,
Options: cloneOptions(b.Options),
References: fileReferenceSourcesToPipeline(b.References),
Validators: b.Validators,
Module: strings.TrimSpace(b.Module),
LLMProfile: strings.TrimSpace(b.LLMProfile),
StructuredOutputRepairAttempts: cloneStructuredOutputRepairAttempts(b.StructuredOutputRepairAttempts),
Retries: b.Retries,
Options: cloneOptions(b.Options),
References: fileReferenceSourcesToPipeline(b.References),
Validators: cloneValidatorOverride(b.Validators),
}
}
func validateStructuredOutputRepairAttemptsNode(node *yaml.Node, context string) error {
if node.Kind != yaml.MappingNode {
return fmt.Errorf("%s must be an object", context)
}
for i := 0; i < len(node.Content); i += 2 {
if node.Content[i].Value != "structured_output_repair_attempts" {
continue
}
if _, err := parseStructuredOutputRepairAttempts(node.Content[i+1], context); err != nil {
return err
}
}
return nil
}
func parseStructuredOutputRepairAttempts(node *yaml.Node, context string) (*int, error) {
if node.Tag == "!!null" {
return nil, fmt.Errorf("%s structured_output_repair_attempts must not be null", context)
}
if node.Kind != yaml.ScalarNode || node.Tag != "!!int" {
return nil, fmt.Errorf("%s structured_output_repair_attempts must be an integer", context)
}
var attempts int
if err := node.Decode(&attempts); err != nil {
return nil, fmt.Errorf("%s structured_output_repair_attempts must be an integer: %w", context, err)
}
if attempts < 0 || attempts > 3 {
return nil, fmt.Errorf("%s structured_output_repair_attempts must be between zero and three", context)
}
return &attempts, nil
}
func cloneStructuredOutputRepairAttempts(attempts *int) *int {
if attempts == nil {
return nil
}
value := *attempts
return &value
}
func LoadFileConfig(path string) (FileConfig, error) {
data, err := os.ReadFile(path)
if err != nil {
@@ -349,6 +402,12 @@ func ParseFileConfigYAML(data []byte) (FileConfig, error) {
if err := decoder.Decode(&fileCfg); err != nil {
return FileConfig{}, fmt.Errorf("decode yaml: %w", err)
}
var trailing any
if err := decoder.Decode(&trailing); err == nil {
return FileConfig{}, fmt.Errorf("config must contain exactly one YAML document")
} else if err != io.EOF {
return FileConfig{}, fmt.Errorf("decode trailing yaml document: %w", err)
}
return fileCfg, nil
}
@@ -511,11 +570,12 @@ func (c *Config) applyFileConfigWithLookup(fileCfg FileConfig, lookup func(strin
return err
}
profile := pipeline.PipelineProfile{
ID: pipelineID,
LLMProfile: llmProfile,
Input: filePipeline.Input.toPipelineBinding(),
Artifacts: make(map[string]pipeline.ArtifactLaneProfile, len(filePipeline.Artifacts)),
References: fileReferenceSourcesToPipeline(filePipeline.References),
ID: pipelineID,
LLMProfile: llmProfile,
StructuredOutputRepairAttempts: cloneStructuredOutputRepairAttempts(filePipeline.StructuredOutputRepairAttempts),
Input: filePipeline.Input.toPipelineBinding(),
Artifacts: make(map[string]pipeline.ArtifactLaneProfile, len(filePipeline.Artifacts)),
References: fileReferenceSourcesToPipeline(filePipeline.References),
}
if filePipeline.Chunk != nil {
profile.Chunk = filePipeline.Chunk.toPipelineBinding()

View File

@@ -50,6 +50,28 @@ func TestFileConfigMinimalVersion4AppliesOverDefaults(t *testing.T) {
}
}
func TestParseFileConfigYAMLRejectsAdditionalDocuments(t *testing.T) {
tests := []struct {
name string
source string
wantErr bool
}{
{name: "trailing whitespace", source: "version: 4\n\n \t", wantErr: false},
{name: "trailing comment", source: "version: 4\n# trailing comment\n", wantErr: false},
{name: "second valid document", source: "version: 4\n---\nversion: 4\n", wantErr: true},
{name: "second empty document", source: "version: 4\n---\n", wantErr: true},
{name: "second malformed document", source: "version: 4\n---\nversion: [\n", wantErr: true},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
_, err := ParseFileConfigYAML([]byte(tt.source))
if (err != nil) != tt.wantErr {
t.Fatalf("ParseFileConfigYAML() error = %v, want error=%t", err, tt.wantErr)
}
})
}
}
func TestFilePipelineLLMProfileIsPresenceAwareAndDetached(t *testing.T) {
const pipelineYAML = `version: 4
pipelines:
@@ -116,6 +138,168 @@ pipelines:
}
}
func TestStructuredOutputRepairAttemptsFileConfigurationPreservesPresenceAndOwnership(t *testing.T) {
const pipelineYAML = `version: 4
pipelines:
main:
%s
input: input
artifacts:
lane:
extract:
module: extract
structured_output_repair_attempts: 2
validators:
- module: validator
structured_output_repair_attempts: 3
`
t.Run("omitted pipeline value remains absent", func(t *testing.T) {
file := parseFileConfig(t, fmt.Sprintf(pipelineYAML, ""))
if got := file.Pipelines["main"].StructuredOutputRepairAttempts; got != nil {
t.Fatalf("file pipeline repair attempts = %v, want nil", *got)
}
cfg := Default()
if err := cfg.ApplyFileConfig(file); err != nil {
t.Fatal(err)
}
if got := cfg.Pipelines["main"].StructuredOutputRepairAttempts; got != nil {
t.Fatalf("pipeline repair attempts = %v, want nil", *got)
}
if got := cfg.Pipelines["main"].Input.StructuredOutputRepairAttempts; got != nil {
t.Fatalf("scalar input binding repair attempts = %v, want nil", *got)
}
})
t.Run("zero is explicit and survives configuration boundaries", func(t *testing.T) {
file := parseFileConfig(t, fmt.Sprintf(pipelineYAML, "structured_output_repair_attempts: 0"))
cfg := Default()
if err := cfg.ApplyFileConfig(file); err != nil {
t.Fatal(err)
}
profile := cfg.Pipelines["main"]
if profile.StructuredOutputRepairAttempts == nil || *profile.StructuredOutputRepairAttempts != 0 {
t.Fatalf("pipeline repair attempts = %v, want explicit zero", profile.StructuredOutputRepairAttempts)
}
lane := profile.Artifacts["lane"]
if lane.Extract.StructuredOutputRepairAttempts == nil || *lane.Extract.StructuredOutputRepairAttempts != 2 {
t.Fatalf("extract repair attempts = %v, want 2", lane.Extract.StructuredOutputRepairAttempts)
}
if lane.Extract.Validators.Validators[0].StructuredOutputRepairAttempts == nil || *lane.Extract.Validators.Validators[0].StructuredOutputRepairAttempts != 3 {
t.Fatalf("validator repair attempts = %v, want 3", lane.Extract.Validators.Validators[0].StructuredOutputRepairAttempts)
}
*file.Pipelines["main"].StructuredOutputRepairAttempts = 1
if got := *cfg.Pipelines["main"].StructuredOutputRepairAttempts; got != 0 {
t.Fatalf("applied config aliased file configuration: got %d, want 0", got)
}
fileProfile := file.Pipelines["main"]
fileLane := fileProfile.Artifacts["lane"]
*fileLane.Extract.StructuredOutputRepairAttempts = 1
*fileLane.Extract.Validators.Validators[0].StructuredOutputRepairAttempts = 1
fileProfile.Artifacts["lane"] = fileLane
file.Pipelines["main"] = fileProfile
configuredLane := cfg.Pipelines["main"].Artifacts["lane"]
if got := *configuredLane.Extract.StructuredOutputRepairAttempts; got != 2 {
t.Fatalf("configured extract aliased file configuration: got %d, want 2", got)
}
if got := *configuredLane.Extract.Validators.Validators[0].StructuredOutputRepairAttempts; got != 3 {
t.Fatalf("configured validator aliased file configuration: got %d, want 3", got)
}
cloned := cloneConfig(cfg)
*cloned.Pipelines["main"].StructuredOutputRepairAttempts = 1
if got := *cfg.Pipelines["main"].StructuredOutputRepairAttempts; got != 0 {
t.Fatalf("cloned config aliased source configuration: got %d, want 0", got)
}
redacted := cfg.Redacted()
*redacted.Pipelines["main"].StructuredOutputRepairAttempts = 1
if got := *cfg.Pipelines["main"].StructuredOutputRepairAttempts; got != 0 {
t.Fatalf("redacted config aliased source configuration: got %d, want 0", got)
}
summary := cfg.RedactedSummaryPayload().(Config)
if summary.Pipelines["main"].StructuredOutputRepairAttempts == nil || *summary.Pipelines["main"].StructuredOutputRepairAttempts != 0 {
t.Fatalf("redacted summary pipeline repair attempts = %v, want explicit zero", summary.Pipelines["main"].StructuredOutputRepairAttempts)
}
data, err := json.Marshal(cfg)
if err != nil {
t.Fatal(err)
}
var roundTripped Config
if err := json.Unmarshal(data, &roundTripped); err != nil {
t.Fatal(err)
}
roundTrippedProfile := roundTripped.Pipelines["main"]
if roundTrippedProfile.StructuredOutputRepairAttempts == nil || *roundTrippedProfile.StructuredOutputRepairAttempts != 0 {
t.Fatalf("round-tripped pipeline repair attempts = %v, want explicit zero", roundTrippedProfile.StructuredOutputRepairAttempts)
}
roundTrippedLane := roundTrippedProfile.Artifacts["lane"]
if roundTrippedLane.Extract.StructuredOutputRepairAttempts == nil || *roundTrippedLane.Extract.StructuredOutputRepairAttempts != 2 {
t.Fatalf("round-tripped extract repair attempts = %v, want 2", roundTrippedLane.Extract.StructuredOutputRepairAttempts)
}
if roundTrippedLane.Extract.Validators.Validators[0].StructuredOutputRepairAttempts == nil || *roundTrippedLane.Extract.Validators.Validators[0].StructuredOutputRepairAttempts != 3 {
t.Fatalf("round-tripped validator repair attempts = %v, want 3", roundTrippedLane.Extract.Validators.Validators[0].StructuredOutputRepairAttempts)
}
})
}
func TestStructuredOutputRepairAttemptsFileConfigurationRejectsInvalidValues(t *testing.T) {
const pipelineYAML = `version: 4
pipelines:
main:
input: input
artifacts:
lane:
extract: extract
%s
`
const bindingYAML = `version: 4
pipelines:
main:
input:
module: input
%s
artifacts:
lane:
extract: extract
`
for _, tt := range []struct {
name string
source string
want string
}{
{name: "null pipeline value", source: "structured_output_repair_attempts: null", want: "pipeline profile structured_output_repair_attempts must not be null"},
{name: "fractional pipeline value", source: "structured_output_repair_attempts: 1.5", want: "pipeline profile structured_output_repair_attempts must be an integer"},
{name: "quoted pipeline value", source: "structured_output_repair_attempts: '1'", want: "pipeline profile structured_output_repair_attempts must be an integer"},
{name: "out of range pipeline value", source: "structured_output_repair_attempts: 4", want: "pipeline profile structured_output_repair_attempts must be between zero and three"},
} {
t.Run(tt.name, func(t *testing.T) {
_, err := ParseFileConfigYAML([]byte(fmt.Sprintf(pipelineYAML, tt.source)))
if err == nil || !strings.Contains(err.Error(), tt.want) {
t.Fatalf("ParseFileConfigYAML() error = %v, want %q", err, tt.want)
}
})
}
for _, tt := range []struct {
name string
source string
want string
}{
{name: "null binding value", source: "structured_output_repair_attempts: null", want: "module binding structured_output_repair_attempts must not be null"},
{name: "noninteger binding value", source: "structured_output_repair_attempts: true", want: "module binding structured_output_repair_attempts must be an integer"},
{name: "out of range binding value", source: "structured_output_repair_attempts: -1", want: "module binding structured_output_repair_attempts must be between zero and three"},
} {
t.Run(tt.name, func(t *testing.T) {
_, err := ParseFileConfigYAML([]byte(fmt.Sprintf(bindingYAML, tt.source)))
if err == nil || !strings.Contains(err.Error(), tt.want) {
t.Fatalf("ParseFileConfigYAML() error = %v, want %q", err, tt.want)
}
})
}
}
func TestFileModuleBindingRejectsExplicitEmptyLLMProfile(t *testing.T) {
const configYAML = `version: 4
pipelines:

View File

@@ -116,6 +116,9 @@ func validatePipelineProfiles(profiles map[string]pipeline.PipelineProfile) erro
if profile.ID != "" && strings.TrimSpace(profile.ID) != id {
return fmt.Errorf("pipeline %q profile id %q does not match map key", id, profile.ID)
}
if err := validateStructuredOutputRepairAttempts(fmt.Sprintf("pipeline %q", id), profile.StructuredOutputRepairAttempts); err != nil {
return err
}
if err := validateBinding(id, "", "input", profile.Input, false); err != nil {
return err
}
@@ -195,6 +198,9 @@ func validateBinding(
binding pipeline.ModuleBinding,
referencesAllowed bool,
) error {
if err := validateStructuredOutputRepairAttempts(referenceContext(pipelineID, laneID, slot), binding.StructuredOutputRepairAttempts); err != nil {
return err
}
if err := validateBindingLLMProfile(pipelineID, laneID, slot, binding); err != nil {
return err
}
@@ -219,6 +225,13 @@ func validateBinding(
return validateReferenceMapForContext(pipelineID, laneID, slot, binding.References, true)
}
func validateStructuredOutputRepairAttempts(context string, attempts *int) error {
if attempts != nil && (*attempts < 0 || *attempts > 3) {
return fmt.Errorf("%s structured_output_repair_attempts must be between zero and three", context)
}
return nil
}
func validateValidatorOverride(pipelineID string, laneID string, slot string, override pipeline.ValidatorOverride) error {
if !override.Set {
return nil
@@ -230,6 +243,9 @@ func validateValidatorOverride(pipelineID string, laneID string, slot string, ov
}
for i, validator := range override.Validators {
context := fmt.Sprintf("%s validators[%d]", referenceContext(pipelineID, laneID, slot), i)
if err := validateStructuredOutputRepairAttempts(context, validator.StructuredOutputRepairAttempts); err != nil {
return err
}
if strings.TrimSpace(validator.Module) == "" {
return fmt.Errorf("%s module must not be empty", context)
}

Some files were not shown because too many files have changed in this diff Show More