Compare commits
148 Commits
906d97b391
...
v0.2.0
| Author | SHA1 | Date | |
|---|---|---|---|
| 39388e96d4 | |||
| 12ac25bd63 | |||
| 394278e1f2 | |||
| 5cd7f8e737 | |||
| bf3fadf9ae | |||
| 58815aaf33 | |||
| ce857966f1 | |||
| a3bd0c1867 | |||
| b05634ee86 | |||
| 4829f94157 | |||
| 67b315099d | |||
| b5c86de4d7 | |||
| 2eeca2ed5a | |||
| b5aaeb1c78 | |||
| 9171b66a41 | |||
| b4363b3b73 | |||
| 241e9d2a89 | |||
| 715fff7b72 | |||
| d627b91b4f | |||
| a67b3aa76d | |||
| a16dcdfa52 | |||
| 46e4466d28 | |||
| 71a004bfc8 | |||
| f8333f2c15 | |||
| f603f7ac64 | |||
| 7a00e7049c | |||
| 2a9db9a957 | |||
| de046a8f13 | |||
| f1a6574013 | |||
| 4bca6d3103 | |||
| 7c569a3d8c | |||
| 8e04ef9e2b | |||
| 53a330587b | |||
| 7cfab8ada0 | |||
| 5c82b62856 | |||
| de8ed41b34 | |||
| d1eaec4dad | |||
| c0ec068f53 | |||
| 5cbd9e56e4 | |||
| 53490cdb59 | |||
| 7c94b5eeed | |||
| 4f2864fc96 | |||
| 893b03fccf | |||
| 256cc98ddb | |||
| e61e522662 | |||
| a4c7eca87b | |||
| 224a8292c4 | |||
| 64d461fc18 | |||
| fb1134e591 | |||
| 1da29e6788 | |||
| 0d947549fb | |||
| 950fba17ce | |||
| 678d2c6099 | |||
| db8db5ffc5 | |||
| 94b3eafb1a | |||
| f6981e2264 | |||
| 2f506f4985 | |||
| fdf8c4afd4 | |||
| 74c793e6a1 | |||
| b5835fbc37 | |||
| fd3f7b85cc | |||
| d86b74f485 | |||
| 59cbf1eb27 | |||
| ee43add75c | |||
| 46761706a2 | |||
| 5a968b64eb | |||
| 7b077c269d | |||
| 80ec939383 | |||
| d3c4d6f133 | |||
| 5ad661f95f | |||
| d63e5c6852 | |||
| fbb8e0d241 | |||
| 8d9a496935 | |||
| d1c48db4bc | |||
| 6bd781d344 | |||
| 8a12c56971 | |||
| 26bd59a5a2 | |||
| 927a7beb88 | |||
| f7059607af | |||
| 63de44c347 | |||
| f0ede9dacc | |||
| 5711f8b9e3 | |||
| da83510234 | |||
| f51b22bea7 | |||
| f320c2fcee | |||
| 4ba1e50a89 | |||
| 2a7e025251 | |||
| 3da20e9d6a | |||
| b1c0faa748 | |||
| 989f2c220b | |||
| 7e35915b3e | |||
| 24238d249e | |||
| 9614469b45 | |||
| 29ee68824d | |||
| aeaaf44ae0 | |||
| d752c51aec | |||
| 8199d95dc1 | |||
| 97cdb01357 | |||
| 7a66095912 | |||
| 9d1356a20e | |||
| e4471fc300 | |||
| 84a2854b5e | |||
| a1b76093ce | |||
| 1aa30a73db | |||
| dc7c0e2f9e | |||
| e2cb0d901a | |||
| 8e0b029f5f | |||
| 9bbf2535dd | |||
| 83fde83a58 | |||
| bef3d1359d | |||
| 1ff449435f | |||
| 6e21c83fd8 | |||
| 6dc9d522b1 | |||
| 8adcf6840d | |||
| f5ed30e455 | |||
| cacf3f24e7 | |||
| f08ca4ddfa | |||
| e1c2f3c202 | |||
| 9614eb540d | |||
| 1b46596a39 | |||
| e043d61a99 | |||
| ad89782c9b | |||
| cd29265d5d | |||
| 2f36b7c3b6 | |||
| b8b3f3abfa | |||
| b490297cde | |||
| 06148074a2 | |||
| 16a998055c | |||
| 97c9a8e5ce | |||
| 66415fd1fa | |||
| bfe25609a7 | |||
| 36e0512454 | |||
| b02f667107 | |||
| ed2b6f4580 | |||
| cb7f145c76 | |||
| 2b9d2eaeaa | |||
| 61016671ab | |||
| b2c076946b | |||
| 250c5c22b8 | |||
| 4b0b166143 | |||
| 90481a0e4b | |||
| 2cbaf20e55 | |||
| a263a0840c | |||
| 14991cf58b | |||
| ab0b4e350c | |||
| 7b2fb0880d | |||
| 748e02db80 | |||
| 23c55f8925 |
2
.gitignore
vendored
2
.gitignore
vendored
@@ -2,6 +2,7 @@
|
||||
notarius
|
||||
notarius-output
|
||||
workspace/
|
||||
.codebase-memory/
|
||||
|
||||
# ---> Go
|
||||
# If you prefer the allow list template instead of the deny list, see community template:
|
||||
@@ -73,4 +74,3 @@ Icon
|
||||
Network Trash Folder
|
||||
Temporary Items
|
||||
.apdisk
|
||||
|
||||
|
||||
63
README.md
63
README.md
@@ -1,34 +1,45 @@
|
||||
# Notarius
|
||||
|
||||
Notarius is a Go CLI for extracting structured artifacts from source material
|
||||
with explicit, configurable pipeline modules.
|
||||
Notarius is a Go CLI for turning source material into structured artifacts with
|
||||
configured extraction pipelines. The implemented D&D workflow reads Seriatim
|
||||
transcript JSON and can produce scene descriptions, item and currency events,
|
||||
NPC identities, combat turns, NPC interactions, and spell casts.
|
||||
|
||||
The current implementation reads Seriatim transcript JSON, chunks the source
|
||||
units, extracts D&D spell-cast artifacts with a Scriptorium-backed LLM runtime,
|
||||
and writes JSON output. Add `--debug` when a per-run inspection bundle is
|
||||
needed.
|
||||
## Quickstart
|
||||
|
||||
```sh
|
||||
OPENROUTER_API_KEY=... \
|
||||
go run ./cmd/notarius run dnd-session \
|
||||
--config examples/dnd-spells.config.yml \
|
||||
Provide an OpenRouter API key through the environment, then run the maintained
|
||||
minimal example:
|
||||
|
||||
~~~
|
||||
OPENROUTER_API_KEY=your-api-key \
|
||||
go run ./cmd/notarius run dnd-session \
|
||||
--config examples/dnd-minimal.config.yml \
|
||||
--input examples/seriatim-minimal-transcript.json
|
||||
```
|
||||
~~~
|
||||
|
||||
This invocation uses the maintained example configuration and input. See the
|
||||
configuration and operations references for profile selection, credentials, and
|
||||
run artifacts.
|
||||
The command publishes a JSON output bundle. Its command syntax and exit
|
||||
behavior are documented in the [CLI reference](docs/cli.md); configuration,
|
||||
credentials, and module selection are owned by the
|
||||
[configuration reference](docs/config.md).
|
||||
|
||||
Useful references:
|
||||
For the complete ordered D&D workflow, use
|
||||
[the complete configuration](examples/dnd-complete.config.yml) with
|
||||
[its synthetic transcript](examples/dnd-complete-transcript.json). It
|
||||
demonstrates all implemented D&D lanes and the supporting campaign references.
|
||||
|
||||
- [CLI reference](docs/cli.md)
|
||||
- [Configuration reference](docs/config.md)
|
||||
- [Operations](docs/operations.md)
|
||||
- [Seriatim input contract](docs/integrations/seriatim.md)
|
||||
- [JSON output contract](docs/integrations/json-output.md)
|
||||
- [D&D spell artifact contract](docs/integrations/dnd-spell-artifacts.md)
|
||||
- [Developer guide](docs/development.md)
|
||||
- [Internal implementation docs](docs/internal/overview.md)
|
||||
- [Maintained example config](examples/dnd-spells.config.yml)
|
||||
- [NPC-grounded example config](examples/dnd-npc-grounded.config.yml)
|
||||
- [Maintained example input](examples/seriatim-minimal-transcript.json)
|
||||
## Documentation
|
||||
|
||||
- [CLI reference](docs/cli.md) — commands, flags, output streams, and exits.
|
||||
- [Configuration reference](docs/config.md) — configuration files, profiles,
|
||||
validation, and module selection.
|
||||
- [Operations](docs/operations.md) — output, state, recovery, and debug
|
||||
handling.
|
||||
- [Integration contracts](docs/integrations/) — Seriatim input and published
|
||||
artifact formats.
|
||||
- [Subprocess consumer guide](docs/consumers/subprocess.md) — invoke Notarius
|
||||
from an orchestrator and consume a published result.
|
||||
- [Internal overview](docs/internal/overview.md) — implemented component map
|
||||
for maintainers.
|
||||
- [Developer guide](docs/development.md) — contributor orientation and
|
||||
validation guidance.
|
||||
- [Future work](docs/roadmap/future.md) — unimplemented ideas and priorities.
|
||||
|
||||
@@ -0,0 +1,98 @@
|
||||
# ADR-0009: Prefer minimal evidence-grounded extraction artifacts
|
||||
|
||||
**Status:** Accepted
|
||||
**Date:** 2026-07-22
|
||||
|
||||
## Context
|
||||
|
||||
Notarius is intended to extract structured facts from source material. Several
|
||||
early D&D artifacts grew to include descriptive prose, inferred relationships,
|
||||
immediate outcomes, summaries, and other enrichment alongside the facts that
|
||||
identify an event or entity. Those fields make one model call responsible for
|
||||
both extraction and synthesis.
|
||||
|
||||
In practice, the richer contracts have produced overlapping or weakly grounded
|
||||
fields and have made structurally valid, semantically coherent output harder for
|
||||
cost-effective smaller models. They also increase prompt size, validation and
|
||||
normalization policy, durable schema surface, downstream coupling, and the
|
||||
number of claims whose provenance must be evaluated.
|
||||
|
||||
The application needs a consistent rule for deciding what belongs in an
|
||||
extractor before redesigning the current D&D spell, NPC, and combat-turn
|
||||
contracts or adding new artifact families.
|
||||
|
||||
## Decision
|
||||
|
||||
An extraction module answers one narrowly stated question and returns the
|
||||
smallest durable structured artifact that usefully answers it.
|
||||
|
||||
Every model-produced field in an extraction artifact must:
|
||||
|
||||
- be necessary to answer the extractor's stated question or serve a known
|
||||
downstream consumer;
|
||||
- represent a fact or bounded classification that can be supported directly by
|
||||
cited source ranges;
|
||||
- remain independently meaningful without model-generated explanatory prose;
|
||||
and
|
||||
- justify the additional prompt, schema, validation, normalization, and
|
||||
compatibility surface it creates.
|
||||
|
||||
Source references are required provenance for extracted records. Auxiliary
|
||||
references may disambiguate identities or canonical names, but they do not
|
||||
establish source facts and are not copied into evidence.
|
||||
|
||||
Extraction artifacts do not include narrative summaries, general analysis,
|
||||
speculative enrichment, inferred biography or relationships, or redundant
|
||||
free-text descriptions by default. When such output has a demonstrated use, it
|
||||
belongs in an explicitly named extraction, classification, enrichment, or
|
||||
analysis module with its own contract and evidence policy.
|
||||
|
||||
Occurrence-level facts are not forced into entity-level attributes. A fact
|
||||
that can change between encounters, such as an NPC's role in a scene, belongs
|
||||
on an occurrence artifact rather than as one scalar property of a normalized
|
||||
NPC registry entry.
|
||||
|
||||
Deterministic mapping and normalization may assign application-owned
|
||||
identifiers, canonicalize known catalog values, order and deduplicate evidence,
|
||||
and collapse records under an explicit identity rule. They must not manufacture
|
||||
removed descriptive fields or synthesize missing claims to satisfy an older
|
||||
contract.
|
||||
|
||||
This is a default design rule, not a prohibition on rich artifacts. A richer
|
||||
field is appropriate when its consumer, evidence semantics, and ownership are
|
||||
explicit.
|
||||
|
||||
## Alternatives considered
|
||||
|
||||
- Keep rich schemas and improve prompts or use larger models. This retains
|
||||
potentially convenient prose but does not resolve overlapping field
|
||||
responsibilities, weak provenance, higher cost, or unnecessary downstream
|
||||
coupling.
|
||||
- Make enrichment fields optional. This reduces rejection pressure but leaves
|
||||
ambiguous artifact semantics and inconsistent records, and many strict
|
||||
structured-output providers still require nullable placeholders.
|
||||
- Keep minimal private LLM schemas while preserving rich durable artifacts.
|
||||
Deterministic code would have to invent, default, or separately derive the
|
||||
missing fields, hiding synthesis behind the extraction boundary.
|
||||
- Use one broad session-analysis module. This reduces the number of lanes but
|
||||
couples unrelated facts, schemas, retries, evaluation, and downstream
|
||||
consumers into one model call.
|
||||
|
||||
## Consequences
|
||||
|
||||
Extraction prompts and response schemas become smaller, more focused, and more
|
||||
suitable for lower-cost models. Artifacts carry fewer unsupported claims, and
|
||||
their evidence and validation policies become easier to explain and evaluate.
|
||||
Independent extractors can evolve, retry, and be consumed without requiring
|
||||
unrelated enrichment.
|
||||
|
||||
Some descriptive convenience fields will disappear from primary artifacts.
|
||||
Consumers that genuinely need them may require a separate module and explicit
|
||||
pipeline step. Entity registries may no longer resolve aliases or relationships
|
||||
unless a dedicated, evidence-grounded capability supplies them.
|
||||
|
||||
Removing durable fields is a schema compatibility change. Each affected
|
||||
artifact requires an explicit version and reference policy; private prompt
|
||||
changes alone are insufficient. Current-behavior integration and internal
|
||||
documentation must change with implementation, while the roadmap owns the
|
||||
proposed contract until then.
|
||||
49
docs/adr/0010-workload-oriented-llm-profile-defaults.md
Normal file
49
docs/adr/0010-workload-oriented-llm-profile-defaults.md
Normal file
@@ -0,0 +1,49 @@
|
||||
# ADR-0010: Use workload-oriented LLM profile defaults
|
||||
|
||||
**Status:** Accepted
|
||||
**Date:** 2026-08-03
|
||||
|
||||
## Context
|
||||
|
||||
LLM-backed D&D operations share an execution-policy choice, but repeating a
|
||||
provider or model-named profile on every module binding ties pipeline structure
|
||||
to a deployment decision. Different environments may require different model,
|
||||
backend, timeout, or reasoning settings while retaining the same workload.
|
||||
|
||||
Notarius also needs a usable default for maintained D&D prompts without making
|
||||
an operator profile mandatory. That default must remain owned by the D&D
|
||||
family, while generic LLM infrastructure stays unaware of domain-specific
|
||||
policy.
|
||||
|
||||
## Decision
|
||||
|
||||
Pipelines may name one workload-oriented default profile, inherited only by
|
||||
selected LLM-backed bindings and validators. Binding-level profile IDs remain
|
||||
intentional exceptions, and the run-wide CLI profile override has highest
|
||||
precedence.
|
||||
|
||||
The D&D family owns an embedded fallback profile named `dnd-extraction`.
|
||||
Operators may provide a complete profile with the same ID through a PromptKit
|
||||
filesystem source. PromptKit selects the higher-precedence matching definition;
|
||||
Notarius does not merge profile documents. Production, development, and local
|
||||
deployments can therefore use different execution policy behind one unchanged
|
||||
pipeline ID.
|
||||
|
||||
## Alternatives considered
|
||||
|
||||
- Repeat a model-named profile on every binding. This makes routine deployment
|
||||
policy changes noisy and obscures the shared workload intent.
|
||||
- Require every deployment to install a profile file. This adds configuration
|
||||
friction and leaves maintained D&D prompts without an application-owned
|
||||
fallback.
|
||||
- Put D&D profile policy in generic LLM infrastructure. This breaks domain
|
||||
ownership and makes generic code depend on one workload.
|
||||
|
||||
## Consequences
|
||||
|
||||
Pipeline configuration expresses workload intent rather than a specific
|
||||
provider or model. Operators can replace the complete execution policy without
|
||||
editing bindings, while binding-level and run-wide exceptions remain available.
|
||||
Profile changes affect resolved pipeline and checkpoint identity, so they may
|
||||
intentionally cause work to be recomputed. The D&D fallback becomes a
|
||||
maintained application execution-policy asset.
|
||||
354
docs/cli.md
354
docs/cli.md
@@ -1,273 +1,165 @@
|
||||
# CLI Reference
|
||||
|
||||
This is the canonical reference for the implemented Notarius command-line
|
||||
interface.
|
||||
interface. For the shortest successful run, see the [README](../README.md).
|
||||
Configuration fields, discovery rules, and selectable module keys are defined
|
||||
in [Configuration](config.md); runtime state and recovery procedures are
|
||||
defined in [Operations](operations.md).
|
||||
|
||||
For the minimal end-to-end invocation, see the [README](../README.md).
|
||||
## Command Summary
|
||||
|
||||
## Commands
|
||||
|
||||
```text
|
||||
~~~
|
||||
notarius help
|
||||
notarius run <pipeline-id> --input path/to/source.json [--config path/to/config.yml] [--only lane-a,lane-b] [--chunk_cache auto|bypass|refresh] [--output-dir path] [--resume] [--recompute-step step-id] [--debug [--debug-dir path]] [--llm-profile id] [--session-id id] [--reference selector=path] [--without-reference selector]
|
||||
notarius run <pipeline-id> --input path/to/source.json [--json] [flags]
|
||||
notarius config validate [--config path/to/config.yml] [--pipeline pipeline-id] [--only lane-a,lane-b]
|
||||
notarius pipelines list [--config path/to/config.yml] [--json]
|
||||
```
|
||||
~~~
|
||||
|
||||
Running `notarius` with no arguments, `notarius help`, `notarius --help`, or
|
||||
`notarius -h` prints usage and exits successfully.
|
||||
Running Notarius without arguments, or with **help**, **--help**, or **-h**,
|
||||
writes the command summary to standard output and exits with status 0.
|
||||
|
||||
## `run`
|
||||
## run
|
||||
|
||||
`notarius run <pipeline-id>` executes a configured pipeline against one input
|
||||
file.
|
||||
~~~
|
||||
notarius run <pipeline-id> --input path/to/source.json [--json] [flags]
|
||||
~~~
|
||||
|
||||
Flags:
|
||||
The **run** command executes the named pipeline for one input file. The
|
||||
pipeline ID and **--input** are required.
|
||||
|
||||
- `--input path`: required source input file.
|
||||
- `--config path`: config file path. If omitted, Notarius uses the discovery
|
||||
rules in [Configuration](config.md#discovery).
|
||||
- `--only lane-a,lane-b`: run only the named artifact lanes. Values are
|
||||
comma-separated and must be non-empty. This retains its existing behavior for
|
||||
implicit single-step pipelines; explicit multi-step pipelines reject it
|
||||
rather than inferring dependency closure.
|
||||
- `--resume`: request checkpoint reuse for this invocation. Checkpoint recording
|
||||
must be enabled in configuration. See
|
||||
[Operations](operations.md#checkpoint-cache) for prerequisites and reuse
|
||||
behavior.
|
||||
- `--recompute-step step-id`: with `--resume` and checkpoint recording enabled,
|
||||
force the named ordered step and every transitive dependent lane to execute.
|
||||
Compatible required predecessors and unrelated lanes remain reusable. The
|
||||
value may identify an explicit step or the implicit single-step ID `default`;
|
||||
it cannot be combined with `--only`.
|
||||
- `--chunk_cache auto|bypass|refresh`: select chunk-plan reuse for this
|
||||
invocation. `auto` reuses a valid plan by canonical source digest, `bypass`
|
||||
performs no plan-cache I/O, and `refresh` regenerates and replaces a valid
|
||||
plan only after chunk validation succeeds. See
|
||||
[Configuration](config.md#state-surfaces) for the persistent setting, precedence,
|
||||
and cache-root selection.
|
||||
- `--output-dir path`: output root. Defaults to `./notarius-output`.
|
||||
- `--debug`: allocate and retain one debug bundle for this invocation.
|
||||
- `--debug-dir path`: debug-bundle root override. This flag requires `--debug`.
|
||||
- `--llm-profile id`: override every effective LLM-capable pipeline module
|
||||
binding with one Scriptorium profile ID. Validator-specific profiles are not
|
||||
overridden.
|
||||
- `--session-id id`: pass a stable prompt session identifier through LLM-backed
|
||||
module calls.
|
||||
- `--reference selector=path`: bind a reference path to a chunk, extractor,
|
||||
merger, or normalizer reference slot. Repeatable.
|
||||
- `--without-reference selector`: remove a configured optional reference binding.
|
||||
Repeatable. It accepts the same selector forms as `--reference`, without
|
||||
`=path`.
|
||||
| Flag | Meaning |
|
||||
| --- | --- |
|
||||
| **--config path** | Use this configuration file. When omitted, configuration discovery applies; see [Configuration](config.md). |
|
||||
| **--input path** | Source input file to process. Required. |
|
||||
| **--output-dir path** | Override the configured output root for this run. |
|
||||
| **--json** | Write the successful run-result receipt as JSON to standard output. |
|
||||
| **--chunk_cache auto\|bypass\|refresh** | Override chunk-plan cache handling for this run. |
|
||||
| **--resume** | Reuse compatible recorded checkpoints when checkpoint recording is enabled. |
|
||||
| **--recompute-step step-id** | With **--resume**, recompute the selected ordered step and its dependent lanes. It cannot be combined with **--only**. |
|
||||
| **--debug** | Retain a debug bundle for this run. |
|
||||
| **--debug-dir path** | Override the debug-bundle root. Requires **--debug**. |
|
||||
| **--only lane-a,lane-b** | Run only the selected comma-separated artifact lanes when that selection is valid for the configured pipeline. |
|
||||
| **--llm-profile id** | Highest-precedence configured profile for selected LLM-backed bindings and validators; it replaces binding and [pipeline](config.md#pipelines) defaults. |
|
||||
| **--session-id id** | Supply a non-empty prompt session identifier to LLM-backed module calls. |
|
||||
| **--reasoning-effort value** | Replace the selected PromptKit profile's reasoning effort for every LLM-backed call in this run. The value must be non-empty and the flag may be specified only once. |
|
||||
| **--clear-reasoning-effort** | Clear reasoning effort inherited from the selected PromptKit profile for every LLM-backed call in this run. |
|
||||
| **--reference selector=path** | Add or replace a file reference binding. Repeatable. |
|
||||
| **--without-reference selector** | Remove a configured optional reference binding. Repeatable. |
|
||||
|
||||
On success, the command prints the completed pipeline ID, normalized output and
|
||||
rejected output counts, and the output directory. A debug-enabled run also
|
||||
prints `debug=<bundle-path>`. If the run completes with warnings, the warning
|
||||
count is printed to stderr.
|
||||
**--chunk_cache** accepts only **auto**, **bypass**, or **refresh**.
|
||||
**--debug-dir**, **--output-dir**, **--session-id**, and
|
||||
**--reasoning-effort**, and **--recompute-step** reject explicit empty values.
|
||||
**--reasoning-effort** and **--clear-reasoning-effort** are mutually exclusive.
|
||||
When neither is present, reasoning effort comes from the selected PromptKit
|
||||
profile. These controls apply to the shared run client, including retries and
|
||||
LLM-backed validators, and do not modify configuration or profile files.
|
||||
Persistent reasoning settings remain a PromptKit profile concern.
|
||||
**--recompute-step** requires **--resume**; checkpoint requirements and reuse
|
||||
behavior are documented in [Operations](operations.md).
|
||||
|
||||
Reference flags are external file bindings resolved against selected chunk,
|
||||
extractor, merger, and normalizer targets before the run starts. Generated
|
||||
artifact bindings are configured in ordered steps and cannot be introduced by a
|
||||
CLI path flag. Flat slot names are accepted only
|
||||
when exactly one selected target declares that slot. For configured reference
|
||||
bindings, precedence, path resolution, and validation, see
|
||||
[Configuration](config.md#pipelines).
|
||||
### Reference selectors
|
||||
|
||||
`--reference` binds or replaces one slot for one selected target. Selectors are:
|
||||
Use **--reference** only for a reference slot declared by the selected
|
||||
configured target. The accepted selector forms are:
|
||||
|
||||
- `slot=path`: valid when exactly one selected target declares `slot`;
|
||||
- `chunk.slot=path`: target the chunker;
|
||||
- `merge.slot=path`: valid when exactly one selected merger declares `slot`;
|
||||
- `lane.slot=path`: valid when exactly one selected extractor, merger, or
|
||||
normalizer in that lane declares `slot`;
|
||||
- `lane.extract.slot=path`: target a lane extractor;
|
||||
- `lane.merge.slot=path`: target a lane merger;
|
||||
- `lane.normalize.slot=path`: target a lane normalizer.
|
||||
| Form | Target |
|
||||
| --- | --- |
|
||||
| slot=path | The unique selected target that declares slot. |
|
||||
| chunk.slot=path | The chunker. |
|
||||
| merge.slot=path | The unique selected merger that declares slot. |
|
||||
| lane.slot=path | The unique extractor, merger, or normalizer in lane that declares slot. |
|
||||
| lane.extract.slot=path | The extractor in lane. |
|
||||
| lane.merge.slot=path | The merger in lane. |
|
||||
| lane.normalize.slot=path | The normalizer in lane. |
|
||||
|
||||
Use `slot=path` when the selected targets declare the slot unambiguously:
|
||||
**--without-reference** uses the same selector forms without =path. Slot
|
||||
names, requiredness, and configured bindings are part of the
|
||||
[configuration contract](config.md).
|
||||
|
||||
```sh
|
||||
go run ./cmd/notarius run dnd-session \
|
||||
--config examples/dnd-spells.config.yml \
|
||||
--input examples/seriatim-minimal-transcript.json \
|
||||
--reference roster=./campaign-roster.txt
|
||||
```
|
||||
### Run output
|
||||
|
||||
Use an explicit selector when multiple selected targets declare the same slot or
|
||||
when you want to target a specific target:
|
||||
Without **--json**, standard output contains the completed pipeline ID, counts
|
||||
of normalized and rejected outputs, and the output directory. A debug-enabled
|
||||
run also prints its debug-bundle path to standard output. A successful run with
|
||||
warnings reports the warning count to standard error. The published JSON bundle
|
||||
is defined by the [JSON output contract](integrations/json-output.md).
|
||||
|
||||
```sh
|
||||
go run ./cmd/notarius run dnd-session \
|
||||
--config examples/dnd-spells.config.yml \
|
||||
--input examples/seriatim-minimal-transcript.json \
|
||||
--reference spells.extract.glossary=./campaign-glossary.txt
|
||||
```
|
||||
With **--json**, successful standard output is exactly one
|
||||
`notarius.run-result.v1` JSON document followed by a newline, with no
|
||||
human-oriented status or debug-path line. Its fields and compatibility policy
|
||||
are defined by the [run-result contract](integrations/run-result.md). A caller
|
||||
must check for exit status 0 before decoding this output; a failed write can
|
||||
leave incomplete standard-output bytes that are not a result document.
|
||||
|
||||
For the maintained NPC-grounded workflow, use the explicit ordered pipeline.
|
||||
The first step produces the normalized NPC artifact; the second step receives
|
||||
it in memory and fans it out to spell extraction, combat extraction, and combat
|
||||
normalization:
|
||||
Example:
|
||||
|
||||
```sh
|
||||
go run ./cmd/notarius run dnd-npc-grounded \
|
||||
--config examples/dnd-npc-grounded.config.yml \
|
||||
--input examples/seriatim-minimal-transcript.json \
|
||||
--output-dir ./npc-grounded-output
|
||||
```
|
||||
~~~
|
||||
OPENROUTER_API_KEY=your-api-key \
|
||||
go run ./cmd/notarius run dnd-session \
|
||||
--config examples/dnd-minimal.config.yml \
|
||||
--input examples/seriatim-minimal-transcript.json
|
||||
~~~
|
||||
|
||||
The generated NPC content remains contextual grounding, not spell or combat
|
||||
evidence. It is represented in manifests and debug summaries by bounded
|
||||
identity and producer provenance, not by payload content or a filesystem path.
|
||||
## config validate
|
||||
|
||||
The same grammar can target chunk, merge, and normalize slots when the configured
|
||||
modules declare them:
|
||||
~~~
|
||||
notarius config validate [--config path/to/config.yml] [--pipeline pipeline-id] [--only lane-a,lane-b]
|
||||
~~~
|
||||
|
||||
```sh
|
||||
go run ./cmd/notarius run dnd-session \
|
||||
--config path/to/config.yml \
|
||||
--input examples/seriatim-minimal-transcript.json \
|
||||
--reference chunk.scene_guide=./campaign-scenes.txt \
|
||||
--reference spells.merge.merge_notes=./merge-notes.txt \
|
||||
--reference spells.normalize.normalization_notes=./normalization-notes.txt
|
||||
```
|
||||
This command loads and validates a configuration. With **--pipeline**, it also
|
||||
resolves that pipeline against the production module catalog. **--only** selects
|
||||
lanes during that resolution and requires **--pipeline**.
|
||||
|
||||
Use `--without-reference` to remove a configured optional binding for a run:
|
||||
|
||||
```sh
|
||||
go run ./cmd/notarius run dnd-session \
|
||||
--config examples/dnd-spells.config.yml \
|
||||
--input examples/seriatim-minimal-transcript.json \
|
||||
--without-reference glossary
|
||||
```
|
||||
|
||||
Use `--session-id` when an external orchestrator needs all prompt calls from one
|
||||
run to share an identifier:
|
||||
|
||||
```sh
|
||||
go run ./cmd/notarius run dnd-session \
|
||||
--config examples/dnd-spells.config.yml \
|
||||
--input examples/seriatim-minimal-transcript.json \
|
||||
--session-id campaign-17-session-04
|
||||
```
|
||||
|
||||
When `cache.checkpoints.enabled` is `true`, runs record checkpoints whether or
|
||||
not `--resume` is present. Add the resume flag to load and reuse compatible
|
||||
recorded work; using it while checkpoint recording is disabled is an error:
|
||||
|
||||
```sh
|
||||
go run ./cmd/notarius run dnd-session \
|
||||
--config examples/dnd-spells.config.yml \
|
||||
--input examples/seriatim-minimal-transcript.json \
|
||||
--resume
|
||||
```
|
||||
|
||||
To selectively rerun one ordered step and its dependent lanes, use the step ID
|
||||
from the configuration. The selected step and dependents are reported as
|
||||
`forced_recompute`; reusable predecessors are reported as `reused`:
|
||||
|
||||
```sh
|
||||
go run ./cmd/notarius run dnd-npc-grounded \
|
||||
--config examples/dnd-npc-grounded.config.yml \
|
||||
--input examples/seriatim-minimal-transcript.json \
|
||||
--resume --recompute-step grounded-events
|
||||
```
|
||||
|
||||
Checkpoint decisions use these categories: `reused`, `executed`,
|
||||
`forced_recompute`, and `dependency_invalidated`. The reason code and bounded
|
||||
detail identify the decision without exposing reference content, local paths,
|
||||
or secrets. `--recompute-step` requires checkpoint recording and `--resume`;
|
||||
unknown step IDs, empty values, and combinations with `--only` are rejected.
|
||||
The operator meanings of checkpoint reason codes are maintained in
|
||||
[Operations](operations.md#resume-and-selective-recompute).
|
||||
|
||||
Use `--debug` to retain the redacted summary and trace bundle for one run. The
|
||||
bundle is allocated before pipeline resolution; once allocated, its path is
|
||||
also printed to stderr if the command fails. Debug-write failures cause exit
|
||||
code `1`.
|
||||
|
||||
```sh
|
||||
go run ./cmd/notarius run dnd-session \
|
||||
--config examples/dnd-spells.config.yml \
|
||||
--input examples/seriatim-minimal-transcript.json \
|
||||
--debug --debug-dir ./notarius-debug
|
||||
```
|
||||
|
||||
Use `refresh` when intentionally replacing the cached plan for the same source:
|
||||
|
||||
```sh
|
||||
go run ./cmd/notarius run dnd-session \
|
||||
--config examples/dnd-spells.config.yml \
|
||||
--input examples/seriatim-minimal-transcript.json \
|
||||
--chunk_cache refresh
|
||||
```
|
||||
|
||||
Use `bypass` for a one-off run that must not inspect or create plan-cache state:
|
||||
|
||||
```sh
|
||||
go run ./cmd/notarius run dnd-session \
|
||||
--config examples/dnd-spells.config.yml \
|
||||
--input examples/seriatim-minimal-transcript.json \
|
||||
--chunk_cache bypass
|
||||
```
|
||||
|
||||
`--diagnostics-dir` has been removed. For checkpoint behavior, durable output,
|
||||
debug-bundle lifecycle, and failure inspection, see [Operations](operations.md).
|
||||
|
||||
## `config validate`
|
||||
|
||||
`notarius config validate` loads and validates configuration.
|
||||
|
||||
Flags:
|
||||
|
||||
- `--config path`: config file path. If omitted, Notarius uses the discovery
|
||||
rules in [Configuration](config.md#discovery).
|
||||
- `--pipeline pipeline-id`: additionally resolve one configured pipeline against
|
||||
the production module catalog.
|
||||
- `--only lane-a,lane-b`: validate resolution for selected artifact lanes. This
|
||||
flag requires `--pipeline`.
|
||||
Success is written to standard output as either config "<path>" is valid or
|
||||
config "<path>" is valid for pipeline "<pipeline-id>".
|
||||
|
||||
Examples:
|
||||
|
||||
```sh
|
||||
~~~
|
||||
go run ./cmd/notarius config validate \
|
||||
--config examples/dnd-spells.config.yml
|
||||
--config examples/dnd-minimal.config.yml \
|
||||
--pipeline dnd-session
|
||||
|
||||
go run ./cmd/notarius config validate \
|
||||
--config examples/dnd-spells.config.yml \
|
||||
--pipeline dnd-session \
|
||||
--only spells
|
||||
```
|
||||
OPENROUTER_API_KEY=validation-placeholder \
|
||||
go run ./cmd/notarius config validate \
|
||||
--config examples/dnd-complete.config.yml \
|
||||
--pipeline dnd-session
|
||||
~~~
|
||||
|
||||
## `pipelines list`
|
||||
The placeholder in the second command is sufficient only for offline
|
||||
validation; it cannot run a provider-backed pipeline.
|
||||
|
||||
`notarius pipelines list` prints configured pipeline IDs in sorted order.
|
||||
## pipelines list
|
||||
|
||||
Flags:
|
||||
~~~
|
||||
notarius pipelines list [--config path/to/config.yml] [--json]
|
||||
~~~
|
||||
|
||||
- `--config path`: config file path. If omitted, Notarius uses the discovery
|
||||
rules in [Configuration](config.md#discovery).
|
||||
- `--json`: print `{"pipelines":[...]}` instead of one ID per line.
|
||||
This command lists configured pipeline IDs in sorted order. By default, it
|
||||
writes one ID per line to standard output. **--json** writes an object shaped as
|
||||
{"pipelines":[...]} instead.
|
||||
|
||||
Examples:
|
||||
|
||||
```sh
|
||||
~~~
|
||||
go run ./cmd/notarius pipelines list \
|
||||
--config examples/dnd-spells.config.yml
|
||||
--config examples/dnd-minimal.config.yml
|
||||
~~~
|
||||
|
||||
go run ./cmd/notarius pipelines list \
|
||||
--config examples/dnd-spells.config.yml \
|
||||
--json
|
||||
```
|
||||
## Output Streams And Exit Statuses
|
||||
|
||||
## Exit Codes
|
||||
Successful commands write their primary result to standard output. Warnings and
|
||||
errors are written to standard error.
|
||||
|
||||
- `0`: command succeeded.
|
||||
- `1`: command syntax was valid, but loading config, resolving modules, running
|
||||
the pipeline, calling the provider, writing output, or writing a requested
|
||||
debug bundle failed.
|
||||
- `2`: command syntax was invalid, a command was unknown, a required argument
|
||||
was missing, or a flag value was malformed.
|
||||
For **run --json**, warnings remain on standard error and standard output is a
|
||||
machine-readable success result only. Syntax and runtime diagnostics remain on
|
||||
standard error. Parse the result only after the process exits with status 0.
|
||||
|
||||
For YAML structure, defaults, Scriptorium profile sources, environment
|
||||
overrides, and selectable module and validator keys, see
|
||||
[Configuration](config.md).
|
||||
| Status | Meaning |
|
||||
| --- | --- |
|
||||
| 0 | The command completed successfully, including root help. |
|
||||
| 1 | Command syntax was valid but configuration loading or validation, pipeline resolution or execution, provider use, output, or requested debug handling failed. |
|
||||
| 2 | The command or flag syntax was invalid, including unknown commands, missing required arguments, invalid flag values, or invalid flag combinations. |
|
||||
|
||||
The root help spellings are the supported help path. Invoking **--help** on
|
||||
**run**, **config validate**, or **pipelines list** is handled by the flag
|
||||
parser as a usage error: it writes an error to standard error and exits with
|
||||
status 2.
|
||||
|
||||
961
docs/config.md
961
docs/config.md
File diff suppressed because it is too large
Load Diff
74
docs/consumers/subprocess.md
Normal file
74
docs/consumers/subprocess.md
Normal file
@@ -0,0 +1,74 @@
|
||||
# Using Notarius As A Subprocess
|
||||
|
||||
Use this workflow when an orchestrator runs Notarius and consumes its published
|
||||
artifacts. The [CLI reference](../cli.md) owns invocation syntax and exit
|
||||
statuses, while the [run-result receipt](../integrations/run-result.md) and
|
||||
[Published JSON Output contract](../integrations/json-output.md) own the
|
||||
durable result formats.
|
||||
|
||||
## Run And Check The Process
|
||||
|
||||
Optionally preflight a selected configuration and pipeline before work starts:
|
||||
|
||||
```sh
|
||||
notarius config validate --config /path/to/notarius.yml --pipeline pipeline-id
|
||||
```
|
||||
|
||||
Invoke the run with explicit paths and machine-readable output. Capture
|
||||
standard output and standard error separately; do not combine them before
|
||||
processing the result.
|
||||
|
||||
```sh
|
||||
notarius run pipeline-id \
|
||||
--config /path/to/notarius.yml \
|
||||
--input /path/to/source.json \
|
||||
--output-dir /path/to/output-root \
|
||||
--json
|
||||
```
|
||||
|
||||
Use absolute paths for supplied input, configuration, output-root, and
|
||||
reference files. When a stable prompt session identifier or references are
|
||||
needed, pass the supported CLI flags. Supply credentials through Notarius's
|
||||
documented configuration and environment mechanisms, never as command-line
|
||||
arguments or generated secret-bearing configuration.
|
||||
|
||||
Wait for the process before interpreting standard output. Only an exit status
|
||||
of 0 permits decoding the receipt. On a nonzero exit, retain standard error for
|
||||
diagnosis and ignore all standard-output bytes: a failed receipt write may have
|
||||
left a partial document.
|
||||
|
||||
## Discover Required Artifacts
|
||||
|
||||
Decode the successful receipt and accept the schema versions supported by the
|
||||
caller. Use its `output_directory` as the bundle root. For the production JSON
|
||||
output, resolve `index_file` under that root with a confinement check and reject
|
||||
an absolute path or a result that escapes the root.
|
||||
|
||||
Read the resulting `index.json` and locate each artifact by `lane_id`, not by a
|
||||
guessed filename. Before decoding a selected payload, verify its descriptor's
|
||||
media type and schema identity against the relevant published artifact
|
||||
contract. The JSON bundle contract links to the available lane contracts.
|
||||
|
||||
If `index.json` has an `evidence_context` descriptor, treat it as a
|
||||
pipeline-wide artifact rather than a lane entry. Verify its six descriptor
|
||||
fields before decoding the linked file according to the [Published Evidence
|
||||
Context contract](../integrations/evidence-context.md). Use each
|
||||
`evidence_refs` entry as the citation to source material. Its surrounding
|
||||
context range and included units explain the citation, but do not widen or
|
||||
replace the cited source reference.
|
||||
|
||||
A zero exit status may still report rejected outputs, warnings, or absent
|
||||
lanes. The caller decides which lane IDs are required for its own work and
|
||||
which are optional; it should make that decision explicitly rather than infer
|
||||
failure from the receipt counts alone.
|
||||
|
||||
## Preserve Provenance And Handle Data Carefully
|
||||
|
||||
Keep the receipt with the published `manifest.json`, and retain
|
||||
`rejected.json` and `warnings.json` when review or later provenance requires
|
||||
them. Treat the input, output bundle, cache, debug bundle, and captured process
|
||||
logs as potentially sensitive data. Apply the caller's access controls and
|
||||
retention policy, and avoid copying secrets into arguments, logs, or
|
||||
provenance records. An evidence-context artifact contains source-unit text and
|
||||
metadata, and selected lanes can cover most of an input; preserve and share it
|
||||
only when that source content is authorized for the recipient.
|
||||
@@ -17,11 +17,13 @@ implemented component map.
|
||||
| Application shape, package boundaries, contracts, dependency direction, runtime guarantees, or safety properties | [Architecture](policy/architecture.md) and relevant [ADRs](adr/) | Architecture defines the intended system and its invariants; ADRs preserve significant decision rationale. |
|
||||
| Any documentation addition or revision | [Documentation Policy](policy/documentation.md) | It defines canonical homes, audiences, current-behavior rules, and maintenance requirements. |
|
||||
| Adding, changing, reviewing, or deleting tests | [Testing Policy](policy/testing.md) | It defines risk-based sufficiency, durable test boundaries, test-double guidance, and criteria for retaining tests. |
|
||||
| CLI composition or command behavior | [CLI Internals](internal/cli.md) and [CLI Reference](cli.md) | The internal guide owns composition and command flow; the reference owns public syntax. |
|
||||
| Building a subprocess caller or changing its result protocol | [Subprocess Consumer Guide](consumers/subprocess.md), [Run Result Receipt](integrations/run-result.md), and [CLI Internals](internal/cli.md) | These separate caller workflow, durable receipt contract, and CLI implementation behavior. |
|
||||
| Configuration loading, resolution, or user-visible configuration behavior | [Configuration Internals](internal/configuration.md) and [Configuration](config.md) | The internal guide owns loading and resolution mechanics; the reference owns the configuration contract. |
|
||||
| Pipeline resolution or execution | [Pipeline Internals](internal/pipeline.md) | It documents profiles, references, validation, retries, checkpoints, and runner behavior. |
|
||||
| Production modules or validators | [Module Internals](internal/modules.md) | It documents implemented module contracts, capabilities, assets, and registration. |
|
||||
| LLM clients, prompts, schemas, profiles, or scheduling | [LLM Runtime](internal/llm.md) | It documents the transport boundary and Scriptorium integration. |
|
||||
| Production modules or validators | [Module Internals](internal/modules.md), [D&D Module Internals](internal/dnd.md), and [D&D integration contracts](integrations/) | The generic guide owns extension mechanics, the D&D guide owns shared family conventions, and the contracts own durable output shapes. |
|
||||
| LLM clients, prompts, schemas, profiles, or scheduling | [LLM Runtime](internal/llm.md) | It documents the transport boundary and PromptKit integration. |
|
||||
| Output, cache, resume, or debug artifacts | [Run State Internals](internal/state.md), [Operations](operations.md), and [Configuration](config.md) | These separate implementation details, operator behavior, and configuration contracts. |
|
||||
| CLI or user-visible configuration behavior | [CLI Reference](cli.md) and [Configuration](config.md) | These are the canonical user and operator references. |
|
||||
| External input formats, artifact schemas, or durable output files | [Integration Contracts](integrations/) | Integration documents define external and durable data contracts. |
|
||||
| Proposed or unimplemented behavior | [Roadmap](roadmap/) | Future work belongs only in roadmap documentation until implemented. |
|
||||
|
||||
|
||||
81
docs/integrations/chunk-map.md
Normal file
81
docs/integrations/chunk-map.md
Normal file
@@ -0,0 +1,81 @@
|
||||
# Accepted Chunk Map
|
||||
|
||||
This document defines the optional durable `chunk-map.json` artifact in a
|
||||
[published JSON bundle](json-output.md). It describes the accepted,
|
||||
materialized chunk plan used by one run. It is not a lane payload and is never
|
||||
an input to a later pipeline step.
|
||||
|
||||
## Contract Identity
|
||||
|
||||
| Property | Value |
|
||||
| --- | --- |
|
||||
| Artifact kind | `source/chunk-map` |
|
||||
| Logical file | `chunk-map.json` |
|
||||
| Media type | `application/json` |
|
||||
| Schema ID | `notarius.source.chunk_map` |
|
||||
| Schema name | `notarius_source_chunk_map_v1` |
|
||||
| Schema version | `v1` |
|
||||
|
||||
The optional `chunk_map` descriptor in `index.json` identifies this artifact.
|
||||
Export is controlled by the JSON output binding described in
|
||||
[Configuration](../config.md#module-bindings-and-validators).
|
||||
|
||||
## Wire Shape
|
||||
|
||||
Every payload has these required fields:
|
||||
|
||||
| Field | Meaning |
|
||||
| --- | --- |
|
||||
| `source_id` | Accepted source-document identity. |
|
||||
| `source_digest` | Lower-case `sha256:` digest of that source document. |
|
||||
| `plan_digest` | Lower-case `sha256:` digest of the logical chunk plan. |
|
||||
| `requested_chunker` | Chunk module selected by the resolved pipeline. |
|
||||
| `producer` | Original accepted-plan producer. `input_module` and `chunk_module` are required; `llm_profile` is optional. |
|
||||
| `plan_annotations` | Plan-level annotation namespace map; `{}` when none are present. |
|
||||
| `chunks` | Non-empty execution-order chunk collection. |
|
||||
|
||||
Each `chunks` entry contains non-empty `id`, zero-based `index`, `source_ref`,
|
||||
positive `unit_count`, and an explicit `annotations` map. `source_ref` contains
|
||||
the same `source_id` as the top-level value plus positive inclusive
|
||||
`start_unit_id` and `end_unit_id` values. Endpoints identify source units; their
|
||||
numeric values do not by themselves establish source-document order.
|
||||
|
||||
Annotation namespaces are non-empty trimmed strings. Their values are arbitrary
|
||||
valid JSON and are retained without interpreting a module-specific namespace.
|
||||
|
||||
## Ordering And Validation
|
||||
|
||||
`chunks` are in execution order. Their indexes are contiguous, start at zero,
|
||||
and equal their array positions; chunk IDs are unique. The emitted map is built
|
||||
only after the selected plan has been accepted and materialized against the
|
||||
source document, so its ranges, unit counts, annotations, and digests describe
|
||||
that exact plan.
|
||||
|
||||
The codec rejects malformed JSON, trailing content, unknown fixed-object
|
||||
fields, invalid identities or digests, invalid annotations, duplicate chunk
|
||||
IDs, non-contiguous indexes, and a `plan_digest` that does not match the
|
||||
reconstructed logical plan. The checked-in
|
||||
[schema](../../internal/framework/chunkmap/assets/schemas/source_chunk_map.v1.json)
|
||||
defines the strict JSON shape.
|
||||
|
||||
## Valid Example
|
||||
|
||||
The compact
|
||||
[source chunk-map fixture](../../internal/framework/chunkmap/testdata/source_chunk_map.v1.json)
|
||||
is decoded by the production codec and demonstrates an accepted map with
|
||||
annotations, producer identity, and ordered chunks.
|
||||
|
||||
## Publication And Compatibility
|
||||
|
||||
The map is present only when a chunk plan was accepted and its export is
|
||||
enabled. It remains publishable if a later lane is rejected, but is absent when
|
||||
chunk-plan validation rejects the plan. `requested_chunker` identifies the
|
||||
current pipeline selection, while `producer` identifies the component that
|
||||
originally produced the accepted plan; they may differ when an accepted plan is
|
||||
reused.
|
||||
|
||||
The map contains structure rather than source content: it excludes transcript
|
||||
bytes, source-unit metadata, chunk text, private model output, reference
|
||||
content, debug data, and filesystem paths. Treat the exported map with the
|
||||
same care as other published output. Publication location and retention are
|
||||
defined in [Operations](../operations.md#output-bundles).
|
||||
@@ -1,10 +1,10 @@
|
||||
# D&D Combat-Turn Artifact Contract
|
||||
# D&D Combat-Turn Artifact
|
||||
|
||||
This document defines the durable artifact, serialization, extraction,
|
||||
candidate-validation, normalization, and production lane boundaries for D&D
|
||||
combat turns.
|
||||
This contract defines the durable combat-action occurrence list produced by
|
||||
`dnd/combat-turns`. It records source-grounded turns and actions; it is not a
|
||||
complete initiative tracker, combat summary, or state model.
|
||||
|
||||
## Artifact identity
|
||||
## Identity and compatibility
|
||||
|
||||
| Property | Value |
|
||||
| --- | --- |
|
||||
@@ -14,165 +14,56 @@ combat turns.
|
||||
| Schema version | `v1` |
|
||||
| Media type | `application/json` |
|
||||
|
||||
The top-level JSON object contains the required `combat_turns` array, which
|
||||
may be empty. Every object rejects unknown fields.
|
||||
`v1` is a strict JSON object with required `combat_turns`; the array may be
|
||||
empty. Turn and source-reference objects reject unknown fields. An incompatible
|
||||
shape change requires a new schema version.
|
||||
|
||||
## JSON shape
|
||||
## Wire shape
|
||||
|
||||
Each combat turn contains these required fields:
|
||||
Each combat turn has these required fields:
|
||||
|
||||
| Field | Shape |
|
||||
| Field | Contract |
|
||||
| --- | --- |
|
||||
| `actor` | Non-empty string. |
|
||||
| `turn_kind` | One of `turn`, `reaction`, `legendary_action`, `lair_action`, or `other`. |
|
||||
| `round` | Required JSON field containing a positive integer or `null`. |
|
||||
| `actions` | Required array with at least one action. |
|
||||
| `summary` | Non-empty string. |
|
||||
| `source_refs` | Required array with at least one source reference. |
|
||||
| `actor` | Non-empty acting character or creature name. |
|
||||
| `turn_kind` | `turn`, `reaction`, `legendary_action`, `lair_action`, or `other`. |
|
||||
| `source_refs` | One or more transcript evidence ranges. |
|
||||
|
||||
Each action contains these required fields:
|
||||
|
||||
| Field | Shape |
|
||||
| --- | --- |
|
||||
| `category` | One of `attack`, `spell`, `movement`, `item`, `ability_check`, `saving_throw`, `condition`, or `other`. |
|
||||
| `declaration` | Non-empty string describing what was declared. |
|
||||
| `targets` | Required array of strings; the array may be empty, but entries may not be empty. |
|
||||
| `resolution` | Required JSON field containing a non-empty string or `null`. |
|
||||
|
||||
Source references use the shared source-reference shape:
|
||||
Each source reference has exactly `source_id`, `start_unit_id`, and
|
||||
`end_unit_id`. It identifies an inclusive current-transcript range; unit IDs
|
||||
are positive and the start may not follow the end.
|
||||
|
||||
```json
|
||||
{
|
||||
"source_id": "session-alpha",
|
||||
"start_unit_id": 1,
|
||||
"end_unit_id": 2
|
||||
"combat_turns": [
|
||||
{
|
||||
"actor": "Mira Thorn",
|
||||
"turn_kind": "turn",
|
||||
"source_refs": [
|
||||
{"source_id": "session-7", "start_unit_id": 31, "end_unit_id": 32}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
`source_id` must be non-empty and both unit IDs must be positive integers. The
|
||||
codec does not resolve references against a source document or enforce source
|
||||
range ordering; those checks belong to the later source-reference validation
|
||||
boundary.
|
||||
## Eligibility, evidence, and normalized form
|
||||
|
||||
## Codec behavior
|
||||
The extractor requires an approved [scene-description artifact](dnd-scene-description-artifacts.md).
|
||||
It emits combat turns only for a chunk with an exact matching scene classified
|
||||
`combat`; an exact non-combat scene produces an accepted empty list. The scene
|
||||
record controls eligibility only: its title, summary, and reference do not
|
||||
become turn evidence. No exact matching scene also produces an empty list and
|
||||
the `scene_classification_unavailable` warning.
|
||||
|
||||
The codec exposes two representations of the same typed artifact:
|
||||
An optional normalized [NPC artifact](dnd-npc-artifacts.md) can ground an
|
||||
actor name. Its registry references are provenance, never combat evidence.
|
||||
Normalization trims and, where possible, canonicalizes actor names; orders and
|
||||
deduplicates exact source references; orders valid-evidence turns by source
|
||||
chronology; and collapses only duplicates with the same actor identity, turn
|
||||
kind, and complete valid evidence. It does not infer turns, initiative, or
|
||||
actions from registry or scene data.
|
||||
|
||||
- Candidate encode/decode preserves invalid enum values, nullable values,
|
||||
required-array presence, required strings, targets, and source references so
|
||||
later validators can report them. Candidate decoding still requires valid
|
||||
JSON, one JSON value, known fields, and the explicitly present `round` and
|
||||
`resolution` keys; `null` is distinct from a missing key.
|
||||
- Approved encode/decode enforces the structural rules in this contract.
|
||||
|
||||
The codec owns the durable JSON Schema, whose object layers all set
|
||||
`additionalProperties` to `false`. Codec metadata contains only
|
||||
`combat_turn_count`.
|
||||
|
||||
The maintained compact fixture is
|
||||
`internal/modules/dnd/codec/combatturns/testdata/dnd_combat_turns.v1.json`.
|
||||
|
||||
## Extraction boundary
|
||||
|
||||
The standalone extractor uses these identities:
|
||||
|
||||
| Property | Value |
|
||||
| --- | --- |
|
||||
| Extractor key | `dnd/combat-turns` |
|
||||
| Capability | `dnd.combat_turns` |
|
||||
| Prompt ID | `dnd.combat_turns` |
|
||||
| Prompt version | `v1` |
|
||||
| Private response-schema key | `dnd_combat_turns_llm` |
|
||||
| Private response-schema ID | `notarius.dnd.combat_turns.llm` |
|
||||
| Default profile | `gemini-2-flash` |
|
||||
|
||||
It requires `chunks` and `source.transcript`, accepts no options, and makes one
|
||||
structured completion for each supplied chunk. The prompt receives the
|
||||
chunk-scoped transcript plus the existing `players`, `party`, and `glossary`
|
||||
inputs, and optionally the deprecated `roster` reference through the shared
|
||||
party mapping. The optional `npcs` reference is an approved normalized NPC
|
||||
artifact used only for identity grounding; it never supplies combat evidence.
|
||||
An external file is validated during preparation. In an ordered pipeline, the
|
||||
same slot may receive the producer's canonical generated artifact at the step
|
||||
handoff.
|
||||
|
||||
The private response envelope has the same fields and JSON types as the durable
|
||||
turn/action shape except that source references contain only `start_unit_id`
|
||||
and `end_unit_id`. It enforces required and nullable field presence, types, and
|
||||
unknown-field rejection, while deterministic validators own enum membership,
|
||||
non-empty values and collections, and positive-number requirements. The
|
||||
extractor assigns the current source ID, removes exact duplicate ranges, and
|
||||
stable-sorts turns by the earliest valid source-document position. Numeric unit
|
||||
IDs are identifiers; source-document slice position determines chronology.
|
||||
Semantically malformed candidate fields remain in the typed result for the
|
||||
configured validation and retry boundary.
|
||||
|
||||
## Deterministic candidate validation
|
||||
|
||||
The standalone validator keys are:
|
||||
|
||||
| Validator | Responsibility |
|
||||
| --- | --- |
|
||||
| `extract/dnd/combat-turns/shape` | Required arrays, strings, nullable fields, positive rounds, and supported enum values. |
|
||||
| `extract/dnd/combat-turns/source_refs` | Source identity, source-unit existence, and range order through the source document. |
|
||||
| `extract/dnd/combat-turns/source_relatedness` | At most one advisory warning per turn when the actor or declared action is not related to cited transcript text. |
|
||||
|
||||
Source-reference and relatedness validators defer malformed shape to the shape
|
||||
validator. Relatedness also defers when any cited source range is invalid. It
|
||||
combines overlapping cited ranges once in document order, compares actors with
|
||||
the shared Unicode-aware NPC identity policy, and checks declaration tokens of
|
||||
at least four Unicode code points against complete cited-text tokens. Targets
|
||||
are not checked deterministically.
|
||||
|
||||
The production D&D registrar exposes the extractor and these validators. Its
|
||||
default extraction chain preserves this order: JSON syntax, private response
|
||||
schema, combat shape, source references, then source relatedness.
|
||||
|
||||
## Normalization boundary
|
||||
|
||||
The standalone normalizer uses key `dnd/combat-turns`, requires `merged`,
|
||||
provides `normalized`, accepts no options, and accepts only the optional
|
||||
structured `npcs` reference. Campaign references are LLM extraction context and
|
||||
are not normalizer inputs. For an external file, the NPC registry is resolved
|
||||
during preparation; for a generated binding, it is resolved at the operation-
|
||||
time handoff. Runtime normalization uses that immutable prepared or handed-off
|
||||
view.
|
||||
|
||||
Normalization policy is `dnd.combat_turns.normalize.v1`. It display-normalizes
|
||||
actor, summary, declarations, targets, and non-null resolutions; canonicalizes
|
||||
exact registry actor and target matches; orders and deduplicates exact source
|
||||
references; stable-sorts records by earliest valid source-document position; and
|
||||
collapses only records with the same actor identity, turn kind, round value, and
|
||||
complete valid evidence set. The first normalized record is retained without
|
||||
merging its actions or prose. Invalid evidence is never eligible for duplicate
|
||||
collapse. Every mutation and collapse emits a bounded warning using the merged
|
||||
input index in its scope.
|
||||
|
||||
The normalizer reports `normalization_policy` and `identity_policy` metadata
|
||||
and fingerprints. An external registry may additionally contribute
|
||||
`npc_registry_digest` and `npc_count`; generated registry identity is retained
|
||||
in framework handoff provenance and dependency fingerprints. The
|
||||
normalized-invariants validator is
|
||||
`normalize/dnd/combat-turns/invariants`; it defers shape and source-reference
|
||||
failures, then checks display normalization, target identity uniqueness,
|
||||
canonical evidence ordering, chronology, and duplicate identity. It rejects
|
||||
with `invalid_combat_turn_normalization` under policy
|
||||
`dnd.combat_turns.validator.normalized.v1`.
|
||||
|
||||
The production D&D registrar exposes the normalizer and normalized-invariants
|
||||
validator. Its default normalization chain is JSON syntax, durable schema,
|
||||
combat shape, normalized invariants, source references, then source
|
||||
relatedness. The lane uses the framework's typed append-order merger and has no
|
||||
merge validator chain.
|
||||
|
||||
## Production manifest and references
|
||||
|
||||
The selectable lane uses extractor and normalizer key `dnd/combat-turns`,
|
||||
`appendorder` for the typed merger, and the durable codec above. A bound `npcs`
|
||||
reference contributes raw-file provenance to the run manifest. A generated
|
||||
binding contributes artifact kind, schema identity, media type, canonical
|
||||
digest, size, and bounded producer provenance. Consumer metadata and checkpoint
|
||||
fingerprints contain no registry names, aliases, content, paths, or NPC source
|
||||
ranges. The normalized lane is emitted as `lanes/<lane-id>.json` by the JSON
|
||||
output module, and warnings and rejection summaries remain in their shared
|
||||
companion files.
|
||||
The [NPC-interaction artifact](dnd-npc-interaction-artifacts.md) records
|
||||
broader NPC occurrences. The [JSON output contract](json-output.md) defines
|
||||
publication, and [D&D module internals](../internal/dnd.md) describes routing
|
||||
and validation mechanics.
|
||||
|
||||
78
docs/integrations/dnd-item-event-artifacts.md
Normal file
78
docs/integrations/dnd-item-event-artifacts.md
Normal file
@@ -0,0 +1,78 @@
|
||||
# D&D Item-Event Artifact
|
||||
|
||||
This contract defines the durable item and currency occurrence list produced by
|
||||
`dnd/item-events`. It records source-grounded discoveries and possession
|
||||
changes; it does not maintain an inventory, balance, or ledger.
|
||||
|
||||
## Identity and compatibility
|
||||
|
||||
| Property | Value |
|
||||
| --- | --- |
|
||||
| Artifact kind | `dnd/item-event-list` |
|
||||
| Schema ID | `notarius.dnd.item_events` |
|
||||
| Schema name | `notarius_dnd_item_events_v1` |
|
||||
| Schema version | `v1` |
|
||||
| Media type | `application/json` |
|
||||
|
||||
`v1` is a strict JSON object with required `events`; the array may be empty.
|
||||
Event and source-reference objects reject unknown fields. An incompatible
|
||||
shape change requires a new schema version.
|
||||
|
||||
## Wire shape
|
||||
|
||||
Every event has required `name`, `kind`, and `source_refs`. `quantity`, `from`,
|
||||
and `to` are optional where the event kind permits them.
|
||||
|
||||
| Field | Contract |
|
||||
| --- | --- |
|
||||
| `name` | Non-empty item or currency display name. |
|
||||
| `kind` | `discovered`, `acquired`, `lost`, `consumed`, or `transferred`. |
|
||||
| `quantity` | Optional positive integer; omit it when no count is established. |
|
||||
| `from` | Optional non-empty losing holder, when allowed by `kind`. |
|
||||
| `to` | Optional non-empty gaining holder, when allowed by `kind`. |
|
||||
| `source_refs` | One or more transcript evidence ranges. |
|
||||
|
||||
Each source reference has exactly `source_id`, `start_unit_id`, and
|
||||
`end_unit_id`. It identifies an inclusive current-transcript range; unit IDs
|
||||
are positive and the start may not follow the end.
|
||||
|
||||
```json
|
||||
{
|
||||
"events": [
|
||||
{
|
||||
"name": "Silver Pieces",
|
||||
"kind": "acquired",
|
||||
"quantity": 20,
|
||||
"to": "party",
|
||||
"source_refs": [
|
||||
{"source_id": "session-7", "start_unit_id": 2, "end_unit_id": 2}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
## Holder rules and minimal extraction
|
||||
|
||||
`discovered` has neither holder; `acquired` requires `to` and forbids `from`;
|
||||
`lost` and `consumed` require `from` and forbid `to`; `transferred` requires
|
||||
both holders. `party` denotes collective possession. A transfer cannot use
|
||||
`party` for either holder and its two normalized holders must differ.
|
||||
|
||||
Only an evidenced discovery or possession change belongs in this artifact.
|
||||
It does not infer quantities or holders, convert currency denominations,
|
||||
calculate balances, or merge nearby events. Campaign references may
|
||||
disambiguate names but are never event evidence. Currency uses the ordinary
|
||||
`name` field and an explicit `quantity` only when the transcript establishes
|
||||
one; each denomination remains a separate event.
|
||||
|
||||
Normalization trims display whitespace, orders and removes exact duplicate
|
||||
source references, then orders events by valid source chronology, name identity
|
||||
and display value, kind, holders, quantity, and reference sequence. It
|
||||
collapses only entries with the same normalized durable fields and complete
|
||||
valid evidence.
|
||||
|
||||
The [JSON output contract](json-output.md) defines publication. See
|
||||
[D&D module internals](../internal/dnd.md) for implementation details and the
|
||||
[NPC-interaction artifact](dnd-npc-interaction-artifacts.md) for a distinct
|
||||
kind of occurrence.
|
||||
@@ -1,127 +1,69 @@
|
||||
# D&D NPC Artifact
|
||||
|
||||
This document defines the durable D&D NPC-list artifact, its JSON codec, and
|
||||
the selectable production NPC pipeline. The normalized JSON payload can be
|
||||
passed explicitly to the spell extractor as an optional caster-name registry
|
||||
or to the combat extractor and normalizer as an actor/target registry. It
|
||||
remains a reference, not spell or combat evidence.
|
||||
This contract defines the durable NPC registry produced by `dnd/npcs`. It is a
|
||||
minimal, source-grounded identity registry for other D&D artifacts, not a
|
||||
character sheet or a relationship summary.
|
||||
|
||||
## Identity
|
||||
## Identity and compatibility
|
||||
|
||||
- Artifact kind: `dnd/npc-list`
|
||||
- Durable schema ID: `notarius.dnd.npcs`
|
||||
- Durable schema name: `notarius_dnd_npcs_v1`
|
||||
- Durable schema version: `v1`
|
||||
- Media type: `application/json`
|
||||
- Identity policy: `dnd.npcs.identity.v1`
|
||||
| Property | Value |
|
||||
| --- | --- |
|
||||
| Artifact kind | `dnd/npc-list` |
|
||||
| Schema ID | `notarius.dnd.npcs` |
|
||||
| Schema name | `notarius_dnd_npcs_v1` |
|
||||
| Schema version | `v1` |
|
||||
| Media type | `application/json` |
|
||||
| Identity policy | `dnd.npcs.identity.v1` |
|
||||
|
||||
The durable JSON Schema is owned by the D&D NPC codec. NPC IDs are derived from
|
||||
the Unicode-normalized, case-folded canonical name using the identity policy.
|
||||
The durable codec enforces the artifact shape and ID syntax; registry identity
|
||||
validation remains a separate deterministic concern.
|
||||
`v1` accepts one strict JSON object with required `npcs`; the array may be
|
||||
empty. NPC and source-reference objects reject unknown fields. An incompatible
|
||||
artifact shape or identity-policy change uses a new version or policy.
|
||||
|
||||
## Output Shape
|
||||
## Wire shape and identity
|
||||
|
||||
The payload is one object with a required top-level `npcs` array:
|
||||
Each NPC has these required fields:
|
||||
|
||||
| Field | Contract |
|
||||
| --- | --- |
|
||||
| `id` | `npc:sha256:` followed by 64 lowercase hexadecimal characters. |
|
||||
| `name` | Non-empty canonical display name. |
|
||||
| `source_refs` | One or more transcript evidence ranges for the identity. |
|
||||
|
||||
A source reference has exactly `source_id`, `start_unit_id`, and `end_unit_id`.
|
||||
The source ID identifies the transcript, unit IDs are positive inclusive unit
|
||||
identifiers, and the start may not follow the end.
|
||||
|
||||
```json
|
||||
{"npcs": []}
|
||||
{
|
||||
"npcs": [
|
||||
{
|
||||
"id": "npc:sha256:99a16589618a04f535a7d21fdcc71a0b1c05d22f752cd492065b1086d97bc3d7",
|
||||
"name": "Mira Thorn",
|
||||
"source_refs": [
|
||||
{"source_id": "session-7", "start_unit_id": 4, "end_unit_id": 5}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
The array may be empty. Every object and nested object rejects unknown fields.
|
||||
The ID is deterministic: normalize the name to Unicode NFKC, normalize the
|
||||
supported apostrophe forms, collapse whitespace, case-fold it, SHA-256 the
|
||||
result, then prefix the lowercase hexadecimal digest with `npc:sha256:`. Each
|
||||
canonical identity and ID appears at most once. Normalization collapses records
|
||||
with the same canonical identity, retains their earliest position, and merges
|
||||
their canonicalized evidence; it does not add aliases, roles, descriptions, or
|
||||
relationship fields.
|
||||
|
||||
## NPC Fields
|
||||
## Scope and consumers
|
||||
|
||||
Each NPC contains exactly these required fields:
|
||||
Only individually identifiable NPC names with transcript evidence belong in
|
||||
this artifact. Groups, generic roles, invented labels, and descriptive
|
||||
enrichment are excluded. Its source references prove registry provenance; they
|
||||
do not become evidence for a spell, interaction, or combat occurrence.
|
||||
|
||||
- `id`: `npc:sha256:` followed by 64 lowercase hexadecimal characters;
|
||||
- `name`: the canonical display name;
|
||||
- `aliases`: an array of alternate display names, which may be empty;
|
||||
- `description`: a concise description;
|
||||
- `relationships`: an array of target/relationship objects, which may be empty;
|
||||
- `source_refs`: at least one source reference supporting the NPC record.
|
||||
|
||||
Each relationship contains required `target` and `relationship` strings. Each
|
||||
source reference contains required `source_id`, `start_unit_id`, and
|
||||
`end_unit_id`; unit IDs are positive integers. Source document identity, unit
|
||||
existence, and range ordering are validated by the source-reference validator
|
||||
when the artifact is used by a pipeline.
|
||||
|
||||
## Codec Boundary
|
||||
|
||||
`EncodeCandidate` and `DecodeCandidate` provide strict single-value JSON
|
||||
serialization while preserving typed values that still need semantic
|
||||
validation. `Encode` and `Decode` are the approved-artifact boundary and
|
||||
require all durable structural fields, non-empty required strings, valid source
|
||||
reference shapes, and the NPC ID pattern.
|
||||
|
||||
Codec metadata contains only `npc_count`. Schema bytes and returned metadata
|
||||
are independent values so callers cannot mutate codec-owned state.
|
||||
|
||||
## Production Pipeline
|
||||
|
||||
The production identities are:
|
||||
|
||||
- extractor: `dnd/npcs`;
|
||||
- artifact kind: `dnd/npc-list`;
|
||||
- normalizer: `dnd/npcs`; and
|
||||
- durable schema: `notarius.dnd.npcs`, version `v1`, media type
|
||||
`application/json`.
|
||||
|
||||
The extractor maps private model records to the current source identity and
|
||||
assigns deterministic IDs. Extraction validation checks shape, source
|
||||
references, and source relatedness. The normalizer then consolidates records
|
||||
by canonical identity or canonical-name/alias matches, preserves the first
|
||||
record's display and output position, unions relationships and exact evidence,
|
||||
rewrites unambiguous relationship targets to canonical names, and validates
|
||||
the retained registry's identity. No LLM is used for consolidation.
|
||||
|
||||
The default extraction chain is `generic/valid_json`,
|
||||
`generic/valid_json_schema`, `extract/dnd/npcs/shape`,
|
||||
`extract/dnd/npcs/source_refs`, and
|
||||
`extract/dnd/npcs/source_relatedness`. The normalize chain adds
|
||||
`normalize/dnd/npcs/identity` before the source-reference and relatedness
|
||||
checks. Relatedness emits bounded warnings when an NPC canonical name or
|
||||
alias is not present near its cited transcript text; opaque campaign
|
||||
references may explain such a warning but do not become evidence.
|
||||
|
||||
## Manifest And Artifact Handoff
|
||||
|
||||
The NPC extractor records prompt and response-schema identities. The durable
|
||||
codec records only `npc_count`; raw names, aliases, descriptions, source
|
||||
references, and payload bytes stay in the lane file rather than manifest
|
||||
metadata. The normalized lane can be consumed by a later ordered step through
|
||||
the registered canonical codec:
|
||||
|
||||
```yaml
|
||||
steps:
|
||||
- id: identify-npcs
|
||||
artifacts:
|
||||
npcs:
|
||||
extract: dnd/npcs
|
||||
normalize: dnd/npcs
|
||||
- id: grounded-events
|
||||
references:
|
||||
npcs:
|
||||
artifact:
|
||||
step: identify-npcs
|
||||
lane: npcs
|
||||
artifacts:
|
||||
spells:
|
||||
extract: dnd/spells
|
||||
normalize: dnd/spells
|
||||
combat:
|
||||
extract: dnd/combat-turns
|
||||
normalize: dnd/combat-turns
|
||||
```
|
||||
|
||||
The framework hands only an accepted normalized artifact across the barrier. It
|
||||
validates the canonical bytes against each consumer slot and clones the
|
||||
operation-time reference for the spell and combat consumers. Generated
|
||||
provenance records the artifact kind, schema identity, media type, canonical
|
||||
digest, size, and producer step/lane/module, but not names, aliases, source
|
||||
ranges, or payload bytes. External normalized files remain supported as
|
||||
explicit references and retain their file provenance.
|
||||
|
||||
NPC source references are registry provenance and are never accepted as spell
|
||||
or combat evidence. Current transcript units remain the only event evidence.
|
||||
This registry can ground actor or caster names in the [spell](dnd-spell-artifacts.md)
|
||||
and [combat-turn](dnd-combat-turn-artifacts.md) artifacts. It is required to
|
||||
resolve the canonical `name` in an [NPC interaction](dnd-npc-interaction-artifacts.md).
|
||||
The [JSON output contract](json-output.md) defines publication, and
|
||||
[D&D module internals](../internal/dnd.md) owns pipeline mechanics.
|
||||
|
||||
78
docs/integrations/dnd-npc-interaction-artifacts.md
Normal file
78
docs/integrations/dnd-npc-interaction-artifacts.md
Normal file
@@ -0,0 +1,78 @@
|
||||
# D&D NPC Interaction Artifact
|
||||
|
||||
This contract defines the durable occurrence list produced by
|
||||
`dnd/npc-interactions`. It records discrete, source-grounded interactions with
|
||||
NPCs already present in a normalized registry; it does not extend that registry
|
||||
or summarize the session.
|
||||
|
||||
## Identity and compatibility
|
||||
|
||||
| Property | Value |
|
||||
| --- | --- |
|
||||
| Artifact kind | `dnd/npc-interaction-list` |
|
||||
| Schema ID | `notarius.dnd.npc_interactions` |
|
||||
| Schema name | `notarius_dnd_npc_interactions_v1` |
|
||||
| Schema version | `v1` |
|
||||
| Media type | `application/json` |
|
||||
|
||||
`v1` is a strict JSON object with required `interactions`; the array may be
|
||||
empty. Interaction and source-reference objects reject unknown fields. An
|
||||
incompatible shape change requires a new schema version.
|
||||
|
||||
## Wire shape
|
||||
|
||||
Each interaction has these required fields:
|
||||
|
||||
| Field | Contract |
|
||||
| --- | --- |
|
||||
| `name` | Non-empty canonical display name from the required NPC registry. |
|
||||
| `kind` | One of the interaction categories below. |
|
||||
| `source_refs` | One or more transcript evidence ranges. |
|
||||
|
||||
Each source reference has exactly `source_id`, `start_unit_id`, and
|
||||
`end_unit_id`. It identifies an inclusive range in the current transcript;
|
||||
unit IDs are positive and the start may not follow the end. Extraction evidence
|
||||
for an interaction is confined to its accepted chunk.
|
||||
|
||||
```json
|
||||
{
|
||||
"interactions": [
|
||||
{
|
||||
"name": "Mira Thorn",
|
||||
"kind": "dialogue",
|
||||
"source_refs": [
|
||||
{"source_id": "session-7", "start_unit_id": 12, "end_unit_id": 13}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
## Interaction categories
|
||||
|
||||
| Kind | Meaning |
|
||||
| --- | --- |
|
||||
| `mentioned` | The NPC is referred to but is not established as present or communicating. |
|
||||
| `noncombat_presence` | The NPC is present and relevant without meaningful dialogue or combat participation. |
|
||||
| `dialogue` | The NPC speaks, responds, or meaningfully participates in a non-combat exchange. |
|
||||
| `combat_ally` | The NPC actively participates in combat on the party's side. |
|
||||
| `combat_opponent` | The NPC actively participates in combat against the party. |
|
||||
| `other` | A clearly evidenced direct occurrence not covered by another category. |
|
||||
|
||||
The categories do not represent motives, relationships, state, or events that
|
||||
the cited transcript does not establish. An `other` entry is not a substitute
|
||||
for uncertain classification.
|
||||
|
||||
## Identity, evidence, and order
|
||||
|
||||
The required normalized [NPC artifact](dnd-npc-artifacts.md) resolves `name`.
|
||||
Registry references are provenance only and never replace an interaction's own
|
||||
evidence. Normalization canonicalizes recognized registry names, orders and
|
||||
deduplicates exact source references, then orders interactions by valid source
|
||||
chronology, NPC comparison identity, display name, kind, and reference sequence.
|
||||
Only entries with the same canonical name, kind, and complete valid evidence
|
||||
sequence are collapsed; distinct categories or evidence remain separate.
|
||||
|
||||
See the [combat-turn artifact](dnd-combat-turn-artifacts.md) for combat-action
|
||||
occurrences and the [JSON output contract](json-output.md) for publication.
|
||||
Pipeline mechanics are described in [D&D module internals](../internal/dnd.md).
|
||||
69
docs/integrations/dnd-scene-description-artifacts.md
Normal file
69
docs/integrations/dnd-scene-description-artifacts.md
Normal file
@@ -0,0 +1,69 @@
|
||||
# D&D Scene-Description Artifact
|
||||
|
||||
This contract defines the durable output of `dnd/scene-descriptions`. Each
|
||||
record classifies one accepted transcript chunk and gives it a minimal
|
||||
source-grounded title and summary.
|
||||
|
||||
## Identity and compatibility
|
||||
|
||||
| Property | Value |
|
||||
| --- | --- |
|
||||
| Artifact kind | `dnd/scene-description-list` |
|
||||
| Schema ID | `notarius.dnd.scene_descriptions` |
|
||||
| Schema name | `notarius_dnd_scene_descriptions_v1` |
|
||||
| Schema version | `v1` |
|
||||
| Media type | `application/json` |
|
||||
|
||||
`v1` is a strict JSON object with required non-empty `scenes`. Scene and
|
||||
source-reference objects reject unknown fields. An incompatible shape change
|
||||
requires a new schema version.
|
||||
|
||||
## Wire shape
|
||||
|
||||
Each scene has exactly these required fields:
|
||||
|
||||
| Field | Contract |
|
||||
| --- | --- |
|
||||
| `id` | Non-empty accepted chunk ID, assigned by Notarius. |
|
||||
| `source_ref` | The assigned inclusive source range for that chunk. |
|
||||
| `kind` | `combat`, `narrative`, `recap`, or `meta`. |
|
||||
| `title` | Non-empty, trimmed, source-grounded title. |
|
||||
| `summary` | Non-empty, trimmed, source-grounded summary. |
|
||||
|
||||
`source_ref` has exactly `source_id`, `start_unit_id`, and `end_unit_id`.
|
||||
Its source ID identifies the input transcript; its positive unit IDs identify
|
||||
the chunk's inclusive range, with the start no later than the end.
|
||||
|
||||
```json
|
||||
{
|
||||
"scenes": [
|
||||
{
|
||||
"id": "chunk-000001",
|
||||
"source_ref": {"source_id": "session-7", "start_unit_id": 1, "end_unit_id": 3},
|
||||
"kind": "narrative",
|
||||
"title": "Arrival at the watchtower",
|
||||
"summary": "The party reaches the ruined watchtower and begins to investigate it."
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
## Meaning and normalized form
|
||||
|
||||
`combat` identifies a chunk where active combat is the central activity.
|
||||
`narrative` is current in-world play that is not principally combat, recap, or
|
||||
meta discussion. `recap` is primarily a recounting of an earlier session, and
|
||||
`meta` is primarily out-of-character discussion. The artifact does not add
|
||||
participants, confidence, events, or information absent from the chunk.
|
||||
|
||||
Normalization trims title and summary, orders scenes by source position and
|
||||
then ID, and removes exact duplicate records. A reused ID with different
|
||||
durable fields, or the same source range with different kind, title, or
|
||||
summary, is invalid. It does not merge adjacent ranges, alter prose, or infer
|
||||
missing scenes.
|
||||
|
||||
The [combat-turn artifact](dnd-combat-turn-artifacts.md) uses an exact matching
|
||||
`combat` scene only as eligibility control; scene title, summary, and source
|
||||
reference never become combat evidence. Publication is defined by the
|
||||
[JSON output contract](json-output.md); implementation details live in
|
||||
[D&D module internals](../internal/dnd.md).
|
||||
@@ -1,211 +1,73 @@
|
||||
# D&D Spell Artifact
|
||||
|
||||
This document is the durable serialized artifact contract for the production
|
||||
D&D spell extractor. Selectable extractor keys are cataloged in
|
||||
[Configuration](../config.md#implemented-production-modules).
|
||||
This contract defines the durable output of the `dnd/spells` extractor and
|
||||
normalizer. It records source-grounded spell-casting occurrences; it is not a
|
||||
spellbook, a rules lookup result, or a record of hypothetical casts.
|
||||
|
||||
## Identity
|
||||
## Identity and compatibility
|
||||
|
||||
- Artifact kind: `dnd/spell-list`
|
||||
- Prompt ID: `dnd.spells`
|
||||
- Response schema key: `dnd_spells`
|
||||
- Response schema ID: `notarius.dnd.spells`
|
||||
- Response schema name: `notarius_dnd_spells_v1`
|
||||
- Response schema version: `v1`
|
||||
- Media type: `application/json`
|
||||
|
||||
The durable JSON Schema is owned by the D&D spell artifact codec. The
|
||||
extractor's private LLM response schema is a separate transport contract: its
|
||||
source-reference objects omit `source_id`, which the extractor assigns while
|
||||
mapping the response to the canonical artifact. The LLM DTO and transport
|
||||
schema are not part of this durable contract.
|
||||
|
||||
The output contains canonical spell casts derived from transcript evidence.
|
||||
Source IDs are assigned from the input identity; source-unit ranges identify
|
||||
the evidence location.
|
||||
|
||||
## Output Shape
|
||||
|
||||
The extractor payload is a JSON object with one required top-level array. Its
|
||||
structure is:
|
||||
|
||||
```text
|
||||
{"spell_casts": [<spell-cast object>, ...]}
|
||||
```
|
||||
|
||||
`spell_casts` must be present. It may be empty when no spell casts are found.
|
||||
When multiple chunk results are combined, spell casts remain in chunk order.
|
||||
When the payload is written as durable output, its logical path is derived from
|
||||
the configured artifact lane ID as defined by the
|
||||
[JSON output contract](json-output.md#output-payload-files).
|
||||
|
||||
## Spell-Cast Fields
|
||||
|
||||
Each spell cast contains exactly these required fields:
|
||||
|
||||
- `caster`: in-world character or creature casting the spell;
|
||||
- `spell`: spell name;
|
||||
- `effect`: concise spell effect in the scene;
|
||||
- `narrative_description`: short description of the spell cast in context;
|
||||
- `source_refs`: transcript source references with extractor-assigned source
|
||||
IDs and evidence unit ranges. It must contain at least one entry.
|
||||
|
||||
All four string fields must be non-empty. `caster` is the canonical in-world
|
||||
caster, not the human player, transcript speaker, or GM when the associated
|
||||
character or creature can be identified. Player and party references may
|
||||
disambiguate that identity, but do not independently establish that a cast
|
||||
occurred. The `spell` value must resolve through the effective SRD-plus-overlay
|
||||
catalog as either a canonical name or alias. Catalog validation accepts aliases
|
||||
but does not rewrite them; unknown fields are rejected.
|
||||
|
||||
`effect` and `narrative_description` record the casting declaration and its
|
||||
immediate resolution as established by the transcript. They do not follow
|
||||
summoned creatures, persistent spell effects, or other downstream consequences
|
||||
through the rest of the scene. They also do not correct the table from
|
||||
published D&D rules or supplement the transcript with model knowledge. When the
|
||||
transcript contains a nonstandard or disputed ruling, the artifact may preserve
|
||||
the immediate observed resolution and attribute relevant reasoning to the GM or
|
||||
table; it must not present that reasoning as a universal game rule. The spell
|
||||
catalog is name-recognition policy, not evidence for spell mechanics or
|
||||
outcomes.
|
||||
|
||||
## Source References
|
||||
|
||||
Each source reference contains exactly three required fields: `source_id`,
|
||||
`start_unit_id`, and `end_unit_id`. The source ID must match the input identity.
|
||||
The unit IDs must be positive integers present in the input, and the start unit
|
||||
must not appear after the end unit. Unknown fields are rejected.
|
||||
|
||||
For each cast, the complete `source_refs` collection identifies the transcript
|
||||
evidence for every factual claim in `caster`, `spell`, `effect`, and
|
||||
`narrative_description`. A cast declaration and its immediate resolution may be
|
||||
cited with separate narrow ranges when intervening units are unrelated. A
|
||||
reported target, roll, amount, condition, interruption, or immediate outcome
|
||||
must be supported by the cited units; otherwise the artifact describes only the
|
||||
supported attempt or declaration. Later behavior by summoned creatures,
|
||||
recurring effects, and other downstream consequences are outside the cast
|
||||
artifact's evidence scope. The deterministic validators establish that ranges
|
||||
are structurally valid and that the spell name is related to cited text.
|
||||
Semantic claim completeness is an extraction policy and remains subject to
|
||||
evaluation rather than deterministic proof.
|
||||
|
||||
Reference slot keys and accepted file types are defined in
|
||||
[Configuration](../config.md#implemented-production-modules). References are
|
||||
supporting disambiguation material, not source evidence, and are not
|
||||
addressable through `source_refs`.
|
||||
|
||||
## Optional NPC Grounding
|
||||
|
||||
The `dnd/spells` extractor accepts an optional `npcs` reference containing one
|
||||
normalized NPC artifact as `application/json`, up to 1 MiB. An external file is
|
||||
validated during preparation; an ordered generated binding is validated at the
|
||||
step handoff. Both paths use the approved NPC codec and identity policy,
|
||||
re-encode canonical durable JSON, and supply that JSON as an operation-time
|
||||
spell prompt input. It helps the model prefer canonical caster names and
|
||||
recognize aliases; it does not establish that a spell was cast.
|
||||
|
||||
NPC source references may identify the run that produced the registry or any
|
||||
other session. They remain registry provenance and are never copied into a
|
||||
spell cast's `source_refs`; every spell evidence range must still identify the
|
||||
current transcript. Generated provenance records producer and canonical
|
||||
artifact identity without payload content or a path. When the slot is absent,
|
||||
the prompt receives exactly `{"npcs":[]}` and the run has no NPC reference
|
||||
provenance or NPC checkpoint fingerprint.
|
||||
|
||||
## Normalization Behavior
|
||||
|
||||
When the `dnd/spells` normalizer is selected, each recognized spell name is
|
||||
rewritten to the effective catalog's canonical display name. Lookup uses the
|
||||
catalog's case-insensitive, whitespace-normalizing, apostrophe-normalizing, and
|
||||
alias rules. Unknown names are preserved exactly for the normalize validators;
|
||||
the normalizer does not guess or apply fuzzy matching.
|
||||
|
||||
Each cast's `source_refs` is copied, sorted by exact `source_id`,
|
||||
`start_unit_id`, and `end_unit_id`, and stripped of exact structural
|
||||
duplicates. Adjacent or overlapping ranges are not merged, and the normalizer
|
||||
does not synthesize references or change their boundaries.
|
||||
|
||||
After those per-cast changes, duplicate identity requires the same canonical
|
||||
spell name, the same caster after case folding and whitespace normalization,
|
||||
and the same complete, non-empty set of source references valid for the source
|
||||
document. Only the first occurrence is retained, in stable order. Its caster,
|
||||
effect, narrative description, and canonical references are preserved without
|
||||
prose merging or source union. Unknown names, empty or invalid evidence, and
|
||||
casts with different evidence remain separate.
|
||||
|
||||
Mutation and duplicate decisions are returned through the normal warnings
|
||||
surface. Warning scopes use the merged input index, such as `spell_casts[0]`,
|
||||
so they remain meaningful even when a later duplicate is removed. The
|
||||
normalizer uses these reason codes:
|
||||
|
||||
| Reason code | Meaning |
|
||||
| Property | Value |
|
||||
| --- | --- |
|
||||
| `spell_name_canonicalized` | A catalog lookup replaced an input name with its canonical display name. |
|
||||
| `spell_name_unresolved` | A name was not found in the effective catalog and was retained unchanged. |
|
||||
| `source_references_normalized` | Reference order changed or exact duplicate references were removed. |
|
||||
| `duplicate_spell_cast_collapsed` | A later cast matched the retained cast's complete duplicate identity. |
|
||||
| Artifact kind | `dnd/spell-list` |
|
||||
| Schema ID | `notarius.dnd.spells` |
|
||||
| Schema name | `notarius_dnd_spells_v1` |
|
||||
| Schema version | `v1` |
|
||||
| Media type | `application/json` |
|
||||
|
||||
Only warnings from an accepted normalize attempt are promoted to
|
||||
`warnings.json`. If an unresolved name reaches the default normalize validator
|
||||
chain, the catalog validator rejects the candidate with `unknown_spell`; the
|
||||
`spell_name_unresolved` warning remains in the attempt's debug artifact. An
|
||||
explicit validator override that accepts the candidate promotes the unresolved
|
||||
warning normally.
|
||||
`v1` is a single strict JSON object. It requires `spell_casts`; the array may
|
||||
be empty. Each spell-cast object and source-reference object rejects unknown
|
||||
fields. An incompatible shape change requires a new schema version.
|
||||
|
||||
## Manifest Metadata
|
||||
## Wire shape
|
||||
|
||||
The extractor adds prompt and response-schema provenance under the artifact lane
|
||||
manifest metadata:
|
||||
Each `spell_casts` entry has these required fields:
|
||||
|
||||
| Field | Contract |
|
||||
| --- | --- |
|
||||
| `caster` | Non-empty in-world character or creature name. |
|
||||
| `spell` | Non-empty spell name. |
|
||||
| `source_refs` | One or more transcript evidence ranges. |
|
||||
|
||||
Every source reference has exactly `source_id`, `start_unit_id`, and
|
||||
`end_unit_id`. The source ID identifies the input transcript; the unit IDs are
|
||||
positive inclusive unit identifiers, and the start may not follow the end in
|
||||
that source. References are evidence for the cast, not campaign-reference or
|
||||
NPC-registry provenance.
|
||||
|
||||
```json
|
||||
{
|
||||
"metadata": {
|
||||
"extractor": {
|
||||
"prompt_id": "dnd.spells",
|
||||
"prompt_version": "v1",
|
||||
"prompt_sha256": "sha256:...",
|
||||
"response_schema_key": "dnd_spells",
|
||||
"response_schema_id": "notarius.dnd.spells",
|
||||
"response_schema_name": "notarius_dnd_spells_v1",
|
||||
"response_schema_version": "v1",
|
||||
"response_schema_sha256": "sha256:...",
|
||||
"catalog_base_id": "dnd-5e-2014-srd-spells",
|
||||
"catalog_digest": "sha256:...",
|
||||
"catalog_overlay_ids": ["campaign.example"],
|
||||
"npc_registry_digest": "sha256:...",
|
||||
"npc_count": 3
|
||||
},
|
||||
"normalizer": {
|
||||
"catalog_base_id": "dnd-5e-2014-srd-spells",
|
||||
"catalog_digest": "sha256:...",
|
||||
"catalog_overlay_ids": ["campaign.example"]
|
||||
}
|
||||
"spell_casts": [
|
||||
{
|
||||
"caster": "Mira Thorn",
|
||||
"spell": "Fireball",
|
||||
"source_refs": [
|
||||
{"source_id": "session-7", "start_unit_id": 12, "end_unit_id": 13}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
`catalog_digest` identifies the effective semantic catalog, while
|
||||
`catalog_overlay_ids` is sorted and empty for a base-only configuration. Raw
|
||||
prompt, schema, catalog, alias, and local overlay-file content are not
|
||||
included in manifest metadata. The `normalizer` metadata uses the same catalog
|
||||
identity fields when that module is selected. Overlay origin, media type, byte
|
||||
size, and raw digest are recorded separately in the manifest's reference
|
||||
provenance; see the [JSON output contract](json-output.md#manifestjson).
|
||||
## Evidence and normalized form
|
||||
|
||||
The `npc_registry_digest` and `npc_count` fields in the example are present for
|
||||
an external NPC registry when the extractor publishes its prepared module
|
||||
metadata. They contain no NPC names, aliases, source references, paths, or raw
|
||||
bytes. A generated registry's identity is instead represented by the framework
|
||||
handoff provenance and dependency fingerprint, so the consumer module metadata
|
||||
does not duplicate it.
|
||||
An entry represents an actual cast or an unambiguous declared attempt. A spell
|
||||
mention, rules discussion, plan, or catalog match alone is not an occurrence.
|
||||
The configured catalog checks the name; it does not establish evidence.
|
||||
|
||||
The extractor's prompt hash, private response-schema hash, and effective catalog
|
||||
digest also contribute independently scoped semantic checkpoint fingerprints.
|
||||
Changing any of those prepared contracts intentionally produces a cold
|
||||
checkpoint miss. Fingerprints contain only digests, never prompt, schema,
|
||||
catalog, or reference content. When an NPC registry is bound, its semantic
|
||||
digest contributes an additional local `npc_registry` fingerprint for an
|
||||
external binding; the manifest metadata contains only that digest and
|
||||
`npc_count`. Raw NPC file provenance remains independently recorded in the
|
||||
manifest's `references` list. Generated bindings contribute the canonical
|
||||
artifact dependency fingerprint and bounded producer provenance instead.
|
||||
When normalization is selected, recognized spell names use the effective
|
||||
catalog's canonical display name. Source references are put in canonical source
|
||||
order and exact duplicate references are removed. A later entry is collapsed
|
||||
only when it has the same canonical spell, the same case- and
|
||||
whitespace-insensitive caster identity, and the same complete valid reference
|
||||
sequence. Remaining entries retain their merged order.
|
||||
|
||||
The optional normalized [NPC artifact](dnd-npc-artifacts.md) can ground a
|
||||
caster name. Its own references remain registry provenance and are never copied
|
||||
into `source_refs`.
|
||||
|
||||
## Related contracts
|
||||
|
||||
The [spell-catalog overlay contract](dnd-spell-catalog-overlays.md) defines
|
||||
the configured catalog additions. The [JSON output contract](json-output.md)
|
||||
defines where this logical artifact is published; [D&D module internals](../internal/dnd.md)
|
||||
describes extraction and validation mechanics.
|
||||
|
||||
@@ -1,19 +1,27 @@
|
||||
# D&D Spell-Catalog Overlay Contract
|
||||
# D&D Spell-Catalog Overlays
|
||||
|
||||
This document defines the JSON format accepted by the D&D spell catalog
|
||||
resolver. An overlay supplies campaign-specific spell names and aliases for
|
||||
recognition. It does not supply spell rules, levels, classes, effects, or
|
||||
source evidence.
|
||||
This document defines the optional JSON overlay consumed by the D&D spell
|
||||
extractor. An overlay contributes campaign spell names and aliases for
|
||||
recognition. It does not define spell rules, effects, levels, classes, or
|
||||
transcript evidence. Bind the optional `spell_catalog` reference as described
|
||||
in [Configuration](../config.md#references-and-ordered-handoffs).
|
||||
|
||||
The `dnd/spells` extractor accepts one optional UTF-8 `application/json` overlay
|
||||
bundle through its `spell_catalog` reference slot. The framework materializes
|
||||
that file relative to the configuration or command-line binding, enforces the
|
||||
1 MiB slot limit, and records its origin and raw digest separately from the
|
||||
effective catalog digest.
|
||||
## Contract Identity
|
||||
|
||||
## Shape
|
||||
| Property | Value |
|
||||
| --- | --- |
|
||||
| Consumer | D&D spell extraction and normalization |
|
||||
| Reference slot | `spell_catalog` |
|
||||
| Media type | `application/json` |
|
||||
| Required schema version | `notarius.dnd.spell-catalog-overlay.v1` |
|
||||
| Base catalog | Embedded D&D 5e 2014 SRD catalog |
|
||||
|
||||
An overlay bundle has this shape:
|
||||
At most one overlay document may be bound. The maintained example is
|
||||
[dnd-spell-catalog.json](../../examples/dnd-spell-catalog.json).
|
||||
|
||||
## Wire Shape
|
||||
|
||||
This is a minimal valid overlay:
|
||||
|
||||
```json
|
||||
{
|
||||
@@ -22,49 +30,43 @@ An overlay bundle has this shape:
|
||||
{
|
||||
"id": "campaign.example",
|
||||
"ruleset": "dnd-5e-2014",
|
||||
"source": {
|
||||
"title": "Example campaign spells",
|
||||
"version": "1",
|
||||
"url": "",
|
||||
"license": ""
|
||||
},
|
||||
"spells": [
|
||||
{
|
||||
"name": "Aegis of Emberfall",
|
||||
"aliases": ["Emberfall Aegis"]
|
||||
}
|
||||
]
|
||||
"source": {"title": "Example campaign spells"},
|
||||
"spells": [{"name": "Aegis of Emberfall"}]
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
The top-level `schema_version` and `catalogs` fields are required. The schema
|
||||
version must be exactly `notarius.dnd.spell-catalog-overlay.v1`, and at least
|
||||
one catalog is required. Catalogs require a unique, non-empty, trimmed `id`,
|
||||
the exact `dnd-5e-2014` `ruleset`, a `source`, and a non-empty `spells` array.
|
||||
| Field | Required | Meaning and constraints |
|
||||
| --- | --- | --- |
|
||||
| `schema_version` | Yes | Exactly `notarius.dnd.spell-catalog-overlay.v1`. |
|
||||
| `catalogs` | Yes | Non-empty array of catalog objects with unique IDs. |
|
||||
| `catalogs[].id` | Yes | Non-empty trimmed string. |
|
||||
| `catalogs[].ruleset` | Yes | Exactly `dnd-5e-2014`. |
|
||||
| `catalogs[].source.title` | Yes | Non-empty trimmed string. |
|
||||
| `catalogs[].source.version` | No | String when present. |
|
||||
| `catalogs[].source.url` | No | String when present. |
|
||||
| `catalogs[].source.license` | No | String when present. |
|
||||
| `catalogs[].spells` | Yes | Non-empty array of spell objects. |
|
||||
| `catalogs[].spells[].name` | Yes | Non-empty trimmed string. |
|
||||
| `catalogs[].spells[].aliases` | No | Array of non-empty trimmed strings when present. |
|
||||
|
||||
`source.title` is required and must be non-empty and trimmed. `source.version`,
|
||||
`source.url`, and `source.license` are optional strings and may be empty.
|
||||
Each spell requires a non-empty, trimmed `name`. `aliases` may be omitted or
|
||||
may be an array of trimmed, non-empty strings; JSON `null` is not an alias
|
||||
array. Overlay objects contain no other supported spell fields.
|
||||
Unknown fields are rejected at every object level. The document must contain
|
||||
one JSON value; `null` is not accepted for optional strings or aliases.
|
||||
|
||||
Decoding is strict: unknown fields, malformed JSON, trailing JSON values, and
|
||||
non-string optional source fields are rejected.
|
||||
## Composition And Compatibility
|
||||
|
||||
## Composition
|
||||
Notarius starts with the embedded base catalog, then applies overlay catalogs
|
||||
in ascending catalog-ID order. A new canonical spell name adds a recognition
|
||||
entry. If an overlay names an existing canonical spell, it augments that spell
|
||||
with aliases while retaining the established display spelling.
|
||||
|
||||
The resolver always starts with the embedded D&D 5e 2014 SRD catalog. Overlay
|
||||
catalogs are sorted by `id` before composition, so the input order does not
|
||||
affect the result. A new canonical name adds a recognition entry. A canonical
|
||||
name matching an existing canonical name augments that spell and keeps the
|
||||
established canonical display spelling. Repeated aliases for the same spell
|
||||
are idempotent.
|
||||
Repeated aliases for the same spell are accepted. A canonical-name, canonical-
|
||||
to-alias, or alias-to-alias collision between different spells is rejected,
|
||||
including a collision with the embedded catalog. Matching uses the catalog’s
|
||||
case, whitespace, and apostrophe normalization, so authors should avoid names
|
||||
or aliases that normalize to another spell.
|
||||
|
||||
Canonical-name display conflicts and canonical/alias or alias/alias collisions
|
||||
between different spells are errors, including collisions with the embedded
|
||||
catalog. Canonical names and aliases use the catalog's case, whitespace, and
|
||||
common-apostrophe normalization rules. The effective catalog returns canonical
|
||||
names in sorted order and produces a semantic SHA-256 digest that is stable
|
||||
under JSON formatting, object-key, catalog, spell, and alias reordering.
|
||||
The overlay is a recognition aid only. The durable spell-artifact schema and
|
||||
source-evidence rules are defined by the
|
||||
[D&D spell artifact contract](dnd-spell-artifacts.md).
|
||||
|
||||
116
docs/integrations/evidence-context.md
Normal file
116
docs/integrations/evidence-context.md
Normal file
@@ -0,0 +1,116 @@
|
||||
# Published Evidence Context
|
||||
|
||||
This contract defines the optional `source/evidence-context` artifact emitted
|
||||
by the production JSON output. Its configuration is owned by
|
||||
[Configuration](../config.md#module-bindings-and-validators); its logical-file
|
||||
discovery is owned by [Published JSON Output](json-output.md).
|
||||
|
||||
## Identity And Discovery
|
||||
|
||||
When enabled, the JSON bundle contains `evidence-context.json` and an
|
||||
`index.json` `evidence_context` descriptor with the same six fields as other
|
||||
pipeline-wide artifact descriptors.
|
||||
|
||||
| Property | Value |
|
||||
| --- | --- |
|
||||
| Artifact kind | `source/evidence-context` |
|
||||
| Media type | `application/json` |
|
||||
| Schema ID | `notarius.source.evidence_context` |
|
||||
| Schema name | `notarius_source_evidence_context_v1` |
|
||||
| Schema version | `v1` |
|
||||
| Logical file | `evidence-context.json` |
|
||||
|
||||
Consumers must discover the file from the descriptor, verify all six descriptor
|
||||
fields, and decode only a supported schema version. The descriptor is optional:
|
||||
its absence means evidence publication was not enabled for that bundle.
|
||||
|
||||
## Payload
|
||||
|
||||
The v1 payload is a JSON object with required `source_id`, `source_digest`,
|
||||
`window_units`, `selected_lanes`, and `contexts` fields. `selected_lanes` and
|
||||
`contexts` are always arrays; an enabled configuration with no accepted direct
|
||||
evidence publishes `contexts: []`.
|
||||
|
||||
```json
|
||||
{
|
||||
"source_id": "session-alpha",
|
||||
"source_digest": "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef",
|
||||
"window_units": 1,
|
||||
"selected_lanes": ["npcs", "spells"],
|
||||
"contexts": [
|
||||
{
|
||||
"context_ref": {
|
||||
"source_id": "session-alpha",
|
||||
"start_unit_id": 10,
|
||||
"end_unit_id": 20
|
||||
},
|
||||
"evidence_refs": [
|
||||
{
|
||||
"lane_id": "spells",
|
||||
"source_ref": {
|
||||
"source_id": "session-alpha",
|
||||
"start_unit_id": 10,
|
||||
"end_unit_id": 10
|
||||
}
|
||||
}
|
||||
],
|
||||
"units": [
|
||||
{
|
||||
"id": 10,
|
||||
"kind": "transcript_segment",
|
||||
"text": "Aria casts Cure Wounds.",
|
||||
"ref": {
|
||||
"source_id": "session-alpha",
|
||||
"start_unit_id": 10,
|
||||
"end_unit_id": 10
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 20,
|
||||
"kind": "transcript_segment",
|
||||
"text": "The party regroups.",
|
||||
"ref": {
|
||||
"source_id": "session-alpha",
|
||||
"start_unit_id": 20,
|
||||
"end_unit_id": 20
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
Each context requires a `context_ref` object and `evidence_refs` and `units`
|
||||
arrays. `context_ref` identifies the first and last included unit. Each
|
||||
evidence entry contains a selected `lane_id` and an original `source_ref`. A
|
||||
unit uses the existing source-unit shape: required `id`, `kind`, `text`, and
|
||||
self `ref`, plus optional JSON-object `metadata`. Fixed payload objects reject
|
||||
unknown fields; unit metadata may contain application-defined JSON values.
|
||||
|
||||
## Citations And Context
|
||||
|
||||
`evidence_refs` are the authoritative citations. They identify the direct
|
||||
references emitted by accepted normalized artifacts. `context_ref` and the
|
||||
units collection include those cited units plus nearby source units selected by
|
||||
the configured window. They are explanatory context, not widened citations.
|
||||
|
||||
Only accepted outputs from the configured lane allowlist contribute. Rejected,
|
||||
failed, absent, and lane-filtered outputs do not contribute. The artifact never
|
||||
contains raw input bytes, prompts, model responses, auxiliary reference
|
||||
content, credentials, or filesystem paths.
|
||||
|
||||
## Ordering And Compatibility
|
||||
|
||||
The selected lane allowlist is lexical. Contexts and units are in source
|
||||
document position order, not numeric unit-ID order. Direct evidence entries
|
||||
are deterministically ordered by lane and source reference. Overlapping or
|
||||
contiguous windows merge, and each source unit appears at most once in the
|
||||
resulting contexts.
|
||||
|
||||
The artifact is additive to the JSON bundle and is not a lane payload,
|
||||
normalized-output count, checkpoint, or generated reference. Consumers that
|
||||
do not need it must tolerate the absent optional descriptor. Consumers that do
|
||||
use it should preserve the artifact and its schema identity with the run
|
||||
provenance, and should treat its source text and metadata as sensitive durable
|
||||
content.
|
||||
@@ -1,191 +1,133 @@
|
||||
# JSON Output
|
||||
# Published JSON Output
|
||||
|
||||
This document is the durable JSON output file-format contract produced by the
|
||||
production JSON encoder and written by the CLI. Selectable output-encoder keys
|
||||
are cataloged in
|
||||
[Configuration](../config.md#implemented-production-modules).
|
||||
This document defines the logical JSON bundle emitted by the production JSON
|
||||
output encoder. The bundle’s physical destination, atomic publication, and
|
||||
retention are operational concerns; see [Operations](../operations.md#output-bundles).
|
||||
Output configuration, including chunk-map and evidence-context publication, belongs in
|
||||
[Configuration](../config.md#module-bindings-and-validators).
|
||||
|
||||
The output module produces the logical bundle described here. The CLI's
|
||||
physical placement and lifecycle for that bundle are defined in
|
||||
[Operations](../operations.md#output-directory).
|
||||
## Bundle Layout
|
||||
|
||||
## Files
|
||||
All paths below are logical, relative, slash-separated bundle paths. The
|
||||
encoder always emits the first four JSON files below and adds lane or
|
||||
pipeline-wide artifact files when their corresponding artifacts are available:
|
||||
|
||||
The encoder writes:
|
||||
A subprocess caller first obtains the physical bundle root from the
|
||||
[run-result receipt](run-result.md), then resolves `index.json` beneath that
|
||||
root for the logical discovery described here.
|
||||
|
||||
- `index.json`
|
||||
- `manifest.json`
|
||||
- `lanes/<lane-id>.json`, one file per normalized serialized artifact
|
||||
- `rejected.json`
|
||||
- `warnings.json`
|
||||
| Path | Purpose |
|
||||
| --- | --- |
|
||||
| `index.json` | Entry point that names the other published files and lane payloads. |
|
||||
| `manifest.json` | Run provenance and result summaries. |
|
||||
| `rejected.json` | Rejected pipeline outputs. |
|
||||
| `warnings.json` | Accepted-output and run warnings. |
|
||||
| `lanes/<safe-lane-id>.json` | One normalized artifact payload for each lane. |
|
||||
| `chunk-map.json` | Optional accepted chunk map, when its export is enabled and available. |
|
||||
| `evidence-context.json` | Optional source-context artifact, when evidence publication is enabled. |
|
||||
|
||||
Files are pretty-printed JSON with a trailing newline when the payload is JSON.
|
||||
Logical file paths are relative, slash-separated, and may not contain `..`.
|
||||
JSON files are pretty-printed with a trailing newline. Lane payloads are
|
||||
accepted only when their media type is `application/json`.
|
||||
|
||||
## `index.json`
|
||||
|
||||
Shape:
|
||||
`index.json` is the bundle’s discovery document. An approved run with no
|
||||
normalized lanes has this valid minimal index:
|
||||
|
||||
```json
|
||||
{
|
||||
"manifest_file": "manifest.json",
|
||||
"output_files": [
|
||||
{
|
||||
"lane_id": "spells",
|
||||
"media_type": "application/json",
|
||||
"file": "lanes/spells.json",
|
||||
"module_key": "noop",
|
||||
"schema_id": "notarius.dnd.spells",
|
||||
"schema_name": "notarius_dnd_spells_v1",
|
||||
"schema_version": "v1"
|
||||
}
|
||||
],
|
||||
"output_files": [],
|
||||
"rejected_file": "rejected.json",
|
||||
"warnings_file": "warnings.json"
|
||||
}
|
||||
```
|
||||
|
||||
`output_files` is sorted by lane ID. Output file names are produced by
|
||||
sanitizing the lane ID:
|
||||
| Field | Required | Meaning |
|
||||
| --- | --- | --- |
|
||||
| `manifest_file` | Yes | Always `manifest.json`. |
|
||||
| `output_files` | Yes | Lane descriptors sorted by `lane_id`. |
|
||||
| `rejected_file` | Yes | Always `rejected.json`. |
|
||||
| `warnings_file` | Yes | Always `warnings.json`. |
|
||||
| `chunk_map` | No | Descriptor for the pipeline-wide `chunk-map.json`; never a lane descriptor. |
|
||||
| `evidence_context` | No | Descriptor for the pipeline-wide `evidence-context.json`; never a lane descriptor. |
|
||||
|
||||
- characters outside `A-Z`, `a-z`, `0-9`, `.`, `_`, and `-` become `_`;
|
||||
- repeated `..` sequences are replaced;
|
||||
- leading and trailing `.`, `_`, and `-` are trimmed;
|
||||
- empty sanitized names are rejected;
|
||||
- two lanes that sanitize to the same output file are rejected.
|
||||
Each lane descriptor has required `lane_id` and `file`. It may also include
|
||||
`media_type`, `module_key`, `schema_id`, `schema_name`, and `schema_version`
|
||||
when supplied by the normalized artifact. Each pipeline-wide artifact
|
||||
descriptor (`chunk_map` or `evidence_context`) contains `artifact_kind`,
|
||||
`file`, `media_type`, `schema_id`, `schema_name`, and `schema_version`. Their
|
||||
payloads are defined by the [Accepted Chunk Map contract](chunk-map.md) and
|
||||
[Published Evidence Context](evidence-context.md), respectively.
|
||||
|
||||
`manifest_file`, `rejected_file`, and `warnings_file` contain the fixed paths
|
||||
shown above. Each `output_files` entry requires `lane_id` and `file`. It also
|
||||
contains the normalized payload `media_type`, normalizer `module_key`, and
|
||||
response `schema_id`, `schema_name`, and `schema_version` when those values are
|
||||
available.
|
||||
The lane path is derived from its lane ID. Characters outside letters, digits,
|
||||
periods, underscores, and hyphens become underscores; `..` sequences are
|
||||
neutralized; leading and trailing periods and underscores are removed. A lane
|
||||
that produces an empty name, or two lanes that produce the same path, makes
|
||||
output encoding fail.
|
||||
|
||||
## Lane Payloads
|
||||
|
||||
Each `lanes/<safe-lane-id>.json` file is the codec-owned normalized JSON for
|
||||
that lane. Consumers should use the index descriptor’s schema identity rather
|
||||
than infer a lane schema from its name. The current D&D payload contracts are
|
||||
[spells](dnd-spell-artifacts.md), [NPCs](dnd-npc-artifacts.md),
|
||||
[NPC interactions](dnd-npc-interaction-artifacts.md),
|
||||
[combat turns](dnd-combat-turn-artifacts.md),
|
||||
[item events](dnd-item-event-artifacts.md), and
|
||||
[scene descriptions](dnd-scene-description-artifacts.md).
|
||||
|
||||
## `manifest.json`
|
||||
|
||||
`manifest.json` contains a run manifest. This abridged example shows its core
|
||||
structure:
|
||||
`manifest.json` is published provenance, not a copy of lane payloads or a
|
||||
checkpoint store. Fields without a value may be omitted. Its top-level fields
|
||||
group into the following externally observable summaries:
|
||||
|
||||
```json
|
||||
{
|
||||
"run_id": "run-123",
|
||||
"pipeline_id": "dnd-session",
|
||||
"artifact_lanes": [
|
||||
{
|
||||
"id": "spells",
|
||||
"extractor": "dnd/spells",
|
||||
"merger": "appendorder",
|
||||
"normalizer": "noop"
|
||||
}
|
||||
],
|
||||
"validation_status": "approved",
|
||||
"started_at": "2026-01-01T00:00:00Z",
|
||||
"completed_at": "2026-01-01T00:00:01Z"
|
||||
}
|
||||
```
|
||||
| Group | Fields |
|
||||
| --- | --- |
|
||||
| Run identity and result | `run_id`, `pipeline_id`, `pipeline_digest`, `schema_version`, `validation_status`, `started_at`, `completed_at` |
|
||||
| Resolved components | `input_module`, `chunker`, `extractors`, `merger`, `normalizer`, `output_encoder`, `artifact_lanes`, `validator_chains`, `module_metadata` |
|
||||
| Source and references | `source_digests`, `references` |
|
||||
| Published result summaries | `normalized_outputs`, `rejected_outputs` |
|
||||
| Execution summaries | `chunk_plan`, `checkpoint_decisions`, `llm_profiles`, `metadata` |
|
||||
|
||||
Fields with empty values may be omitted by JSON encoding.
|
||||
`references` records provenance such as the target, slot, origin, digest,
|
||||
media type, size, and generated-artifact identity. It does not contain
|
||||
reference content. `normalized_outputs` and `rejected_outputs` likewise
|
||||
summarize results without embedding lane payload bytes. A chunk-plan summary is
|
||||
provenance for the plan used by this run; cache records, debug artifacts, and
|
||||
other operational state are not published as bundle files.
|
||||
|
||||
The manifest fields are:
|
||||
Each `llm_profiles` entry identifies effective, non-secret LLM execution
|
||||
provenance:
|
||||
|
||||
- `run_id`, `pipeline_id`, and `pipeline_digest`: run and resolved-pipeline
|
||||
identity;
|
||||
- `input_module`, `chunker`, `extractors`, `merger`, `normalizer`, and
|
||||
`output_encoder`: resolved module keys;
|
||||
- `chunk_plan`: payload-free provenance for the effective chunk plan. `mode`
|
||||
is the effective cache mode; `action` is `reused`, `generated`,
|
||||
`refreshed`, or `bypassed` when a plan was materialized. `requested_module`
|
||||
is the current pipeline chunker, while `producer_input_module`,
|
||||
`producer_module`, `producer_llm_profile`, `producer_references`,
|
||||
`producer_metadata`, `source_digest`, `plan_digest`, `plan_schema_version`,
|
||||
and `created_at` describe the stored or generated producer when available.
|
||||
A cached plan can therefore identify a producer different from the requested
|
||||
module. This object never embeds ranges, units, annotations, prompts,
|
||||
responses, or reference content;
|
||||
- `module_metadata` and `artifact_lanes`: module and per-lane provenance,
|
||||
including prompt and response-schema provenance when provided;
|
||||
- `validator_chains`: resolved validation points and validators;
|
||||
- `source_digests` and `references`: source and reference provenance;
|
||||
- `normalized_outputs` and `rejected_outputs`: payload-free result summaries;
|
||||
- `llm_profiles`: selected profile IDs and provider or model names when
|
||||
available;
|
||||
- `metadata`: the effective prompt `session_id`;
|
||||
- `validation_status`: `approved` or `rejected`;
|
||||
- `started_at` and `completed_at`: UTC run timestamps.
|
||||
| Field | Required | Meaning |
|
||||
| --- | --- | --- |
|
||||
| `id` | Yes | Selected PromptKit profile identifier. |
|
||||
| `provider` | No | Notarius adapter provider identifier. |
|
||||
| `model` | No | Effective provider model identifier. |
|
||||
| `backend_id` | No | Effective PromptKit backend registration identifier. Endpoint-only profiles omit it. |
|
||||
| `reasoning_effort` | No | Effective opaque provider reasoning setting. An empty or explicitly cleared setting is omitted. |
|
||||
|
||||
`source_digests` contains source document digests only. Bound references are
|
||||
recorded separately under `references`, which contains provenance only: target
|
||||
stage, lane ID when present, slot name, origin type and URI, digest, media
|
||||
type, byte size, and binding source. Reference content is not written to
|
||||
durable output.
|
||||
These values describe observed execution; they are not a backend-registration
|
||||
interface. Entries that differ by backend or effective reasoning remain
|
||||
distinct even when their profile, provider, and model are otherwise equal.
|
||||
|
||||
Reference `stage` is `chunk`, `extract`, `merge`, or `normalize`. `lane_id` is
|
||||
omitted for chunk references and present for extract, merge, and normalize
|
||||
references.
|
||||
## Rejections And Warnings
|
||||
|
||||
`validation_status` is `approved` when no outputs were rejected and `rejected`
|
||||
when one or more outputs were rejected.
|
||||
`rejected.json` is always an object with a `rejected` array. Each entry has
|
||||
required `stage` and `message`; `step_id`, `lane_id`, `module_key`, `chunk_id`,
|
||||
`chunk_index`, `validator_name`, `reason_code`, `attempt_count`, and
|
||||
`diagnostic_artifact_path` are present only when applicable.
|
||||
|
||||
Producer warnings and the current run's chunk-validation warnings remain in
|
||||
`warnings.json`. The manifest records only provenance and decision summaries;
|
||||
empty producer-only values are omitted for compatibility with existing readers.
|
||||
`warnings.json` is always an object with a `warnings` array. Each warning has
|
||||
`reason_code` and `message`; `scope` is optional. Both arrays are empty when
|
||||
there is nothing to report.
|
||||
|
||||
`validator_chains` records the resolved validator chain for each validation
|
||||
point. Entries include stage, lane ID when applicable, module key, and validators
|
||||
with key and execution class. Empty chains are recorded with an empty
|
||||
`validators` array, including chains resolved from explicit empty config
|
||||
overrides.
|
||||
## Compatibility
|
||||
|
||||
`normalized_outputs` summarizes each normalized lane output without embedding
|
||||
payload bytes. Entries include lane ID, normalizer module key, source ID, media
|
||||
type, and response schema provenance where available.
|
||||
|
||||
`rejected_outputs` summarizes rejected module outputs without embedding raw
|
||||
payload bytes. Entries include stage, lane, module, chunk, validator or reason,
|
||||
message, attempt count, and optional diagnostic artifact path.
|
||||
|
||||
## Output Payload Files
|
||||
|
||||
Each normalized serialized artifact is written to
|
||||
`lanes/<sanitized-lane-id>.json`. The JSON output encoder is domain-neutral and
|
||||
accepts only artifacts whose codec media type is `application/json`. The file
|
||||
contains the codec-owned JSON bytes pretty-printed.
|
||||
|
||||
The schema of each lane payload is owned by that artifact contract. For the
|
||||
current D&D lanes, see [D&D Spell Artifact](dnd-spell-artifacts.md),
|
||||
[D&D NPC Artifact](dnd-npc-artifacts.md), and
|
||||
[D&D Combat-Turn Artifact](dnd-combat-turn-artifacts.md).
|
||||
|
||||
## `rejected.json`
|
||||
|
||||
Shape:
|
||||
|
||||
```json
|
||||
{
|
||||
"rejected": []
|
||||
}
|
||||
```
|
||||
|
||||
When output validation rejects an output, each entry contains `stage` and
|
||||
`message`. It includes `lane_id`, `module_key`, `chunk_id`, `chunk_index`,
|
||||
`validator_name`, `reason_code`, `attempt_count`, and
|
||||
`diagnostic_artifact_path` when applicable.
|
||||
|
||||
## `warnings.json`
|
||||
|
||||
Shape:
|
||||
|
||||
```json
|
||||
{
|
||||
"warnings": [
|
||||
{
|
||||
"scope": "extract",
|
||||
"reason_code": "example",
|
||||
"message": "human-readable warning"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
`warnings` is an empty array when no warnings are reported.
|
||||
Each warning requires `reason_code` and `message`; `scope` is omitted when it is
|
||||
empty.
|
||||
The index is the authoritative map from a logical lane to its published
|
||||
payload. Consumers must tolerate omitted optional manifest and descriptor
|
||||
fields, and should rely on the linked artifact contract for each lane’s JSON
|
||||
shape. This contract describes the published logical bundle only; it does not
|
||||
promise a filesystem layout or expose internal state formats.
|
||||
|
||||
106
docs/integrations/pkg-promptkit.md
Normal file
106
docs/integrations/pkg-promptkit.md
Normal file
@@ -0,0 +1,106 @@
|
||||
# PromptKit Integration
|
||||
|
||||
Notarius pins
|
||||
[`gitea.maximumdirect.net/eric/promptkit` v0.5.0](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0)
|
||||
as its in-process prompt engine. The upstream
|
||||
[Go package consumer guide](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0/docs/consumers/pkg-promptkit.md)
|
||||
owns the public engine API, and the upstream
|
||||
[format reference](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0/docs/formats.md)
|
||||
owns prompt, profile, and schema file contracts.
|
||||
|
||||
## Supported Boundary
|
||||
|
||||
Notarius relies on the root `promptkit` package to:
|
||||
|
||||
- construct an `Engine` with filesystem-backed prompt, schema, and optional
|
||||
operator and application-fallback profile sources;
|
||||
- prepare one frozen execution from a `RunRequest` with named inline artifacts,
|
||||
variables, a direct session ID, prompt identity, and profile selection, then
|
||||
record credential-redacted details and run that exact execution;
|
||||
- return rendered debug material, validated structured output, selected
|
||||
profile, backend, effective model metadata, and token usage;
|
||||
- register the optional conventional `local` backend through `BackendLocal`,
|
||||
`LocalBackend`, and `WithBackend`;
|
||||
- distinguish structured-output validation failure from execution failure; and
|
||||
- identify a missing explicit profile through `ErrProfileNotFound` and backend
|
||||
admission exhaustion through `ErrCapacityExceeded`.
|
||||
|
||||
The pinned
|
||||
[`BackendLocal`, `LocalBackend`, and `WithBackend` API](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0/backends.go)
|
||||
owns the registration and backend-capacity contract.
|
||||
|
||||
For one completion, the adapter calls `PrepareExecution`, takes a
|
||||
caller-owned `Details` snapshot, and calls `RunPrepared` for that same opaque
|
||||
prepared execution. It defers `Discard` for every unexecuted handle. Explicit
|
||||
profile preflight uses `Engine.InspectProfile`; it does not prepare a synthetic
|
||||
prompt. PromptKit's prepared handle, inspection result, and capacity-error
|
||||
types stay inside the Notarius LLM adapter.
|
||||
|
||||
When a PromptKit profile and runtime override leave `temperature`, `max_tokens`,
|
||||
or `top_p` unset, Notarius leaves that control unset as well. Compatible
|
||||
providers therefore apply their own defaults; an operator that requires a
|
||||
specific sampling value must select it explicitly in the profile or runtime
|
||||
override.
|
||||
|
||||
Notarius does not use PromptKit's optional `ArtifactReader`. It materializes
|
||||
source and reference content itself and supplies owned inline artifacts at the
|
||||
adapter boundary. It also retains responsibility for pipeline retries,
|
||||
scheduling, debug persistence, redaction, profile provenance, and conversion
|
||||
from private model responses into durable domain artifacts.
|
||||
|
||||
Notarius sends its trimmed run session through PromptKit's direct session
|
||||
field, which is authoritative for provider session behavior. It also retains
|
||||
the same value as the `session_id` prompt variable for maintained prompt
|
||||
compatibility. Session IDs are stable, non-secret correlation identifiers and
|
||||
may be exposed to providers and provider observability.
|
||||
|
||||
Notarius records PromptKit's selected backend ID and effective reasoning
|
||||
setting as optional run-manifest provenance. Endpoint-only profiles have no
|
||||
backend ID. Debug prompt material also retains the selected backend ID and
|
||||
PromptKit's stable lower-case `effective_model_params` JSON, which may include
|
||||
`backend_id`. Notarius production configuration exposes one optional
|
||||
conventional `local` registration. It does not expose a general user-defined
|
||||
PromptKit backend registry. Endpoint-only profiles remain supported unchanged.
|
||||
|
||||
Notarius retains its application-wide scheduled client around the PromptKit
|
||||
adapter. PromptKit may apply a narrower limit for the selected backend;
|
||||
endpoint-only profiles have no such backend limit. The adapter translates
|
||||
PromptKit capacity rejection into the provider-neutral Notarius
|
||||
`ErrLLMCapacityExceeded` contract. It may include the normalized selected
|
||||
backend ID in safe diagnostic context, without exposing PromptKit's capacity
|
||||
error type, and leaves retries to the calling pipeline stage.
|
||||
|
||||
## Profile Sources And Compatibility
|
||||
|
||||
Notarius gives PromptKit the configured operator profile source, registered
|
||||
application fallback profile assets, and optional backend registration through
|
||||
the same construction path for inspection and execution. PromptKit owns the
|
||||
resulting source precedence and strict profile parsing: a matching operator
|
||||
profile is a complete replacement for a fallback or built-in profile, while an
|
||||
invalid matching document fails instead of falling through. The operator
|
||||
configuration and deployment workflow are defined in
|
||||
[Configuration](../config.md#promptkit-profiles) and
|
||||
[Operations](../operations.md#promptkit-profile-deployment).
|
||||
|
||||
Notarius supports this boundary against PromptKit v0.5.0. Its fallback source,
|
||||
prepared-execution, inspection, and typed capacity APIs are used as public
|
||||
upstream contracts; other PromptKit APIs or file-format behavior are not
|
||||
implicitly supported. A dependency upgrade requires reviewing the adapter,
|
||||
profile-source construction, and this compatibility statement against the
|
||||
pinned upstream documentation.
|
||||
|
||||
## Notarius Ownership
|
||||
|
||||
[LLM Runtime Internals](../internal/llm.md) describes how Notarius mounts
|
||||
module assets, maps its transport-neutral completion contract, prepares and
|
||||
executes requests, validates output, records provenance, captures debug
|
||||
material, redacts errors, and preserves timeout ownership.
|
||||
[D&D Module Internals](../internal/dnd.md) owns the embedded
|
||||
`dnd-extraction` fallback profile and the maintained D&D prompt defaults.
|
||||
[Configuration](../config.md#promptkit-profiles) defines how a Notarius
|
||||
configuration selects one PromptKit profile source and optionally registers
|
||||
the conventional local backend.
|
||||
|
||||
PromptKit API or format changes outside this boundary are not implicitly
|
||||
supported. Updating the pinned version requires reviewing the adapter and
|
||||
profile/configuration contracts against the upstream documentation.
|
||||
68
docs/integrations/run-result.md
Normal file
68
docs/integrations/run-result.md
Normal file
@@ -0,0 +1,68 @@
|
||||
# Run Result Receipt
|
||||
|
||||
`notarius run --json` writes this receipt to standard output when a run
|
||||
completes successfully. It lets a subprocess caller discover the physical root
|
||||
of the published output bundle without parsing interactive command output.
|
||||
Command syntax, streams, and exit statuses are defined in the
|
||||
[CLI reference](../cli.md); logical files within the bundle are defined in the
|
||||
[Published JSON Output contract](json-output.md).
|
||||
|
||||
## Schema
|
||||
|
||||
The current schema version is `notarius.run-result.v1`.
|
||||
|
||||
| Field | Required | Meaning |
|
||||
| --- | --- | --- |
|
||||
| `schema_version` | Yes | Exactly `notarius.run-result.v1`. |
|
||||
| `run_id` | Yes | The finalized Notarius run identifier. |
|
||||
| `pipeline_id` | Yes | The effective pipeline identifier. |
|
||||
| `output_directory` | Yes | Absolute path to the published, run-specific output bundle. |
|
||||
| `index_file` | For the production JSON output | Logical path `index.json`; omitted for other output modules. |
|
||||
| `normalized_output_count` | Yes | Number of final normalized outputs. |
|
||||
| `rejected_output_count` | Yes | Number of recorded rejected outputs. |
|
||||
| `warning_count` | Yes | Number of final run warnings. |
|
||||
| `validation_status` | Yes | The final run manifest validation status. |
|
||||
| `debug_directory` | No | Absolute path to the run-specific debug bundle when requested debug capture completed. |
|
||||
|
||||
For the production `json` output module, `index_file` is present only when the
|
||||
completed run returned exactly one logical output file named `index.json`.
|
||||
For another output module, its absence does not indicate a failed run.
|
||||
|
||||
```json
|
||||
{
|
||||
"schema_version": "notarius.run-result.v1",
|
||||
"run_id": "run-1770000000000000000-0123456789abcdef0123456789abcdef",
|
||||
"pipeline_id": "dnd-session",
|
||||
"output_directory": "/work/results/run-1770000000000000000-0123456789abcdef0123456789abcdef",
|
||||
"index_file": "index.json",
|
||||
"normalized_output_count": 6,
|
||||
"rejected_output_count": 2,
|
||||
"warning_count": 1,
|
||||
"validation_status": "rejected"
|
||||
}
|
||||
```
|
||||
|
||||
## Paths And Bundle Discovery
|
||||
|
||||
`output_directory` and `debug_directory`, when present, are lexical absolute
|
||||
paths. They identify the paths used by Notarius and do not resolve symlinks.
|
||||
`output_directory` is the run-specific bundle, not the configured output root.
|
||||
|
||||
The receipt is a summary and discovery document. It does not contain lane
|
||||
descriptors, payloads, manifest data, rejections, warnings, or file contents.
|
||||
For the production JSON output, resolve `index_file` beneath
|
||||
`output_directory`, reject path escapes, and use the
|
||||
[Published JSON Output contract](json-output.md) to discover logical files and
|
||||
lane payloads.
|
||||
|
||||
## Delivery And Compatibility
|
||||
|
||||
Notarius writes the receipt only after the output bundle has been published and
|
||||
any requested debug terminal reporting has completed. Standard output is not
|
||||
transactional: a result-write failure returns a nonzero status and can leave
|
||||
partial bytes. Consumers must ignore standard output unless the process exits
|
||||
with status 0.
|
||||
|
||||
Future versions may add optional fields to this schema. Consumers must tolerate
|
||||
unknown fields. An incompatible field or semantic change requires a new
|
||||
`schema_version` value.
|
||||
@@ -1,69 +1,73 @@
|
||||
# Seriatim Transcript JSON
|
||||
# Seriatim Transcript Input
|
||||
|
||||
This document is the external input contract consumed by the production
|
||||
Seriatim input adapter. Selectable input-adapter keys are cataloged in
|
||||
[Configuration](../config.md#implemented-production-modules).
|
||||
This document defines the JSON transcript accepted by the production Seriatim
|
||||
input adapter. It is a source input, not a durable lane artifact. Configure the
|
||||
input adapter through [Configuration](../config.md#production-module-keys).
|
||||
|
||||
## Adapter
|
||||
## Contract Identity
|
||||
|
||||
- Source format: `application/vnd.seriatim+json`
|
||||
| Property | Value |
|
||||
| --- | --- |
|
||||
| Consumer | Seriatim input adapter |
|
||||
| Media type | `application/vnd.seriatim+json` |
|
||||
| Source document kind | `transcript` |
|
||||
| Source-unit kind | `transcript_segment` |
|
||||
|
||||
## Accepted Shape
|
||||
|
||||
The input must be one JSON object with top-level `metadata` and `segments`
|
||||
fields. This covers the maintained minimal fixture and Seriatim intermediate
|
||||
output that provides the same required segment fields.
|
||||
The input is one JSON object containing `metadata` and a non-empty `segments`
|
||||
array. This minimal document is valid:
|
||||
|
||||
The maintained example is
|
||||
[examples/seriatim-minimal-transcript.json](../../examples/seriatim-minimal-transcript.json).
|
||||
```json
|
||||
{
|
||||
"metadata": {"id": "session-alpha"},
|
||||
"segments": [
|
||||
{
|
||||
"id": 1,
|
||||
"start": 0,
|
||||
"end": 4,
|
||||
"speaker": "Aria",
|
||||
"text": "Aria casts Cure Wounds."
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
Required top-level fields:
|
||||
The maintained two-segment input is
|
||||
[seriatim-minimal-transcript.json](../../examples/seriatim-minimal-transcript.json).
|
||||
|
||||
- `metadata`: an object. Its entries are accepted as source metadata.
|
||||
- `segments`: a non-empty array of segment objects.
|
||||
| Field | Required | Meaning and constraints |
|
||||
| --- | --- | --- |
|
||||
| `metadata` | Yes | JSON object. Its entries become source metadata; no particular metadata key is otherwise required. |
|
||||
| `segments` | Yes | Non-empty array of segment objects, kept in input order. |
|
||||
| `segments[].id` | Yes | Positive canonical decimal integer, supplied as a JSON number or string. IDs must be unique. |
|
||||
| `segments[].start` | Yes | Finite, non-negative numeric value, supplied as a JSON number or string. |
|
||||
| `segments[].end` | Yes | Finite, non-negative numeric value that is not earlier than `start`. |
|
||||
| `segments[].speaker` | Yes | String that is non-empty after trimming. |
|
||||
| `segments[].text` | Yes | String that is non-empty after trimming. Its original text is retained. |
|
||||
|
||||
Required segment fields:
|
||||
Additional top-level and segment fields are ignored. A missing required field,
|
||||
`null` in place of an object or array, malformed JSON, or more than one
|
||||
top-level JSON value is rejected.
|
||||
|
||||
- `id`: a positive integer JSON number or canonical decimal string without
|
||||
leading zeros or surrounding whitespace;
|
||||
- `start`: a finite, non-negative JSON number or numeric string;
|
||||
- `end`: a finite, non-negative JSON number or numeric string that is not less
|
||||
than `start`;
|
||||
- `speaker`: a non-empty string;
|
||||
- `text`: a non-empty string.
|
||||
## Source Identity And References
|
||||
|
||||
Other top-level and segment fields, such as `categories`, are ignored.
|
||||
The adapter chooses the source ID in this order:
|
||||
|
||||
Multiple top-level JSON values are rejected.
|
||||
1. a non-empty source ID supplied by the calling request;
|
||||
2. non-empty string `metadata.id`;
|
||||
3. non-empty string `metadata.source_id`;
|
||||
4. `seriatim:` followed by the first 16 hexadecimal characters of the raw
|
||||
input’s SHA-256 digest.
|
||||
|
||||
## Validation
|
||||
Each accepted segment becomes one source unit whose unit ID is `segments[].id`.
|
||||
Its self-reference uses the derived source ID and the same segment ID for both
|
||||
range endpoints. Artifact contracts use those segment IDs when they cite
|
||||
transcript evidence.
|
||||
|
||||
The adapter rejects empty input, malformed JSON, multiple top-level JSON values,
|
||||
non-object segment values, duplicate segment IDs, and any violation of the
|
||||
shape or field constraints above.
|
||||
## Compatibility
|
||||
|
||||
Segment text is preserved as provided, but it must not be empty after trimming.
|
||||
|
||||
## Derived Identity
|
||||
|
||||
Notarius identifies the parsed source in this order:
|
||||
|
||||
1. `metadata.id`, when it is a non-empty string after trimming;
|
||||
2. `metadata.source_id`, when it is a non-empty string after trimming;
|
||||
3. `seriatim:<first-16-hex-chars-of-raw-sha256>`.
|
||||
|
||||
The exact raw input SHA-256 remains the basis of the fallback source ID. The
|
||||
source digest recorded in output provenance is instead the SHA-256 of the
|
||||
canonical generic source document, excluding the digest field itself. It covers
|
||||
the derived source identity, document kind and format, ordered units and their
|
||||
self-references, and accepted metadata. Segment IDs become the unit IDs used by
|
||||
artifact source references; each produced unit carries a self-reference whose
|
||||
source ID is the derived document ID and whose start and end IDs both equal the
|
||||
segment ID.
|
||||
|
||||
## Compatibility Limit
|
||||
|
||||
This contract covers only Seriatim transcript JSON with the top-level
|
||||
`metadata` object and `segments` array described here. Broader Seriatim output
|
||||
schemas are compatible only when they provide these required fields with the
|
||||
accepted types.
|
||||
This adapter accepts only the shape described here. A broader Seriatim export
|
||||
is usable only when it supplies this object, metadata, and segment shape with
|
||||
the stated types and constraints. Unknown additional fields do not add
|
||||
Notarius behavior.
|
||||
|
||||
163
docs/internal/cli.md
Normal file
163
docs/internal/cli.md
Normal file
@@ -0,0 +1,163 @@
|
||||
# CLI Internals
|
||||
|
||||
This document describes **internal/cli**, Notarius's production composition
|
||||
root. The [CLI reference](../cli.md) owns command syntax and exit statuses;
|
||||
[Configuration](../config.md) owns configuration values; and
|
||||
[Operations](../operations.md) owns filesystem layout, recovery, and operator
|
||||
procedures.
|
||||
|
||||
## Inputs, Outputs, And Boundaries
|
||||
|
||||
The CLI accepts process arguments, standard streams, and injectable options
|
||||
used by tests and embedding code. It writes command results to the supplied
|
||||
streams and returns a process exit status. For a run, it also creates the
|
||||
production catalog and runtime collaborators, hands a prepared pipeline and
|
||||
source bytes to the framework, and places the logical files returned by the
|
||||
runner.
|
||||
|
||||
It is the only boundary allowed to compose concrete registries, LLM clients,
|
||||
cache/checkpoint collaborators, debug recorders, and physical output paths.
|
||||
Pipeline modules receive interfaces and request data rather than CLI streams or
|
||||
filesystem roots. The [Architecture](../policy/architecture.md) defines this
|
||||
composition-root boundary; [Pipeline Internals](pipeline.md) owns resolution,
|
||||
preparation, and runner mechanics after their inputs are supplied.
|
||||
|
||||
## Dispatch And Configuration Handoff
|
||||
|
||||
The root dispatcher handles help, configuration validation, pipeline listing,
|
||||
and a pipeline run. It normalizes injectable options before dispatch so that a
|
||||
missing production dependency fails as a command error rather than reaching
|
||||
execution.
|
||||
|
||||
Commands that need configuration use one shared loader. The CLI discovers the
|
||||
file, parses it through **internal/core/config**, starts from defaults, applies
|
||||
the file and supported environment overrides, and then validates it for the
|
||||
command. The configured discovery and precedence contract is in
|
||||
[Configuration](../config.md), while the loading and resolution mechanics are
|
||||
in [Configuration Internals](configuration.md).
|
||||
|
||||
Configuration validation without a selected pipeline checks structural
|
||||
configuration only. Validation with a selected pipeline also builds the
|
||||
effective catalog, resolves the pipeline, and verifies every explicit effective
|
||||
PromptKit profile. Selected LLM-backed input, chunk, lane, output, and validator
|
||||
profiles are inspected
|
||||
against the configured PromptKit source and backend registrations without
|
||||
loading a prompt or performing generation, so an unknown or invalid profile
|
||||
fails before pipeline preparation. Credential availability remains an
|
||||
execution-time concern. Pipeline listing validates configuration before
|
||||
returning normalized, sorted identifiers.
|
||||
|
||||
## Production Composition
|
||||
|
||||
The production composition helper allocates every framework registry and the
|
||||
prompt-asset registry, then registers the generic, Seriatim, and D&D module
|
||||
families in that order. The resulting registries provide both the module
|
||||
catalog used for resolution and the concrete constructors used for preparation.
|
||||
Tests may provide a catalog or registries instead; production code must not
|
||||
silently merge an injected partial catalog with production registrations.
|
||||
|
||||
The production LLM factory builds one PromptKit-backed client from the resolved
|
||||
**promptkit.profile_dir** or **promptkit.profile_file** source, attaches the
|
||||
profile-provenance recorder, creates one scheduler from the effective global
|
||||
LLM limit, and wraps the client before it reaches modules. Registration and LLM
|
||||
construction errors are returned before a pipeline is prepared. Configuration
|
||||
field definitions remain in [Configuration](../config.md#promptkit-profiles);
|
||||
the D&D registrar's fallback profile assets and the adapter mechanics remain in
|
||||
[LLM Runtime](llm.md).
|
||||
|
||||
The factory also accepts `LLMRuntimeOverrides`, whose reasoning pointer
|
||||
preserves inherit, replace, and clear states across the composition boundary.
|
||||
Run orchestration constructs this value from the mutually exclusive
|
||||
`--reasoning-effort` and `--clear-reasoning-effort` controls. Absence preserves
|
||||
a nil pointer, replacement is trimmed, and clear uses a non-nil empty string.
|
||||
The same override reaches the one shared production client, checkpoint
|
||||
identity, and debug invocation metadata. Persistent reasoning configuration
|
||||
remains owned by PromptKit profiles; Notarius configuration has no reasoning
|
||||
field.
|
||||
|
||||
## Run Orchestration
|
||||
|
||||
After parsing and validating a run invocation, the CLI performs this ordered
|
||||
handoff:
|
||||
|
||||
1. load and validate configuration, then apply command-level operational
|
||||
overrides;
|
||||
2. create and validate a safe run identity, then allocate a debug bundle only
|
||||
when requested;
|
||||
3. build the effective catalog, resolve requested reference changes, resolve
|
||||
the effective pipeline, and inspect its explicit effective PromptKit
|
||||
profiles;
|
||||
4. materialize external or generated references and record redacted invocation
|
||||
and resolution provenance when debug capture is enabled;
|
||||
5. construct registries, the scheduled LLM client, prepared modules, and the
|
||||
requested cache/checkpoint collaborators;
|
||||
6. read the source input and invoke the framework runner; and
|
||||
7. write the runner's logical output files only after a successful run, then
|
||||
complete the command report and user-facing result.
|
||||
|
||||
Preparation happens before source parsing, so module construction and
|
||||
dependency failures cannot begin stage execution. The CLI also preserves the
|
||||
framework's result and warning information when it writes summaries and the
|
||||
final command result. Detailed state lifecycle, resume handling, and physical
|
||||
path confinement are maintained in [Run State Internals](state.md) and
|
||||
[Operations](../operations.md).
|
||||
|
||||
For `run --json`, the CLI constructs and encodes its private run-result receipt
|
||||
after a successful runner result is available, before it publishes logical
|
||||
output files. It writes the prepared receipt to standard output only after
|
||||
output publication and requested debug terminalization succeed. A receipt-write
|
||||
failure exits with runtime status 1 and may leave partial standard-output bytes,
|
||||
but the already-published output bundle remains complete and requested debug
|
||||
reporting remains successfully terminalized. The CLI reports a bounded
|
||||
command-owned error and does not repeat terminal reporting. The receipt remains
|
||||
a CLI reporting concern rather than a framework or output-module responsibility;
|
||||
its public contract is the
|
||||
[run-result receipt](../integrations/run-result.md).
|
||||
|
||||
## Failure Mapping And Terminal Reporting
|
||||
|
||||
Argument, flag, and invocation-combination failures are reported to standard
|
||||
error before runtime composition and use the syntax error class. Once an
|
||||
invocation is syntactically valid, configuration loading and validation,
|
||||
resolution, registration, profile checks, reference materialization, module
|
||||
construction, input reads, runner failures, output publication, and requested
|
||||
debug handling use the runtime failure class. The public status numbers and
|
||||
stream contract are defined in the [CLI reference](../cli.md#output-streams-and-exit-statuses).
|
||||
|
||||
When debug capture has been allocated, one command-state value records the
|
||||
known run result. Guarded terminalization writes a success report once, or
|
||||
attempts a failure report and error record once. A persistence failure is
|
||||
reported in addition to the original failure and never replaces it. If a debug
|
||||
path exists, failure output includes that path so the retained diagnostic data
|
||||
is discoverable.
|
||||
|
||||
## Invariants To Preserve
|
||||
|
||||
- Only the CLI composes production implementations and physical runtime roots.
|
||||
- Configuration and resolved composition failures occur before module
|
||||
preparation or source parsing.
|
||||
- A runner's logical files are published only after a successful run.
|
||||
- Production registries and a caller-supplied catalog or registries are
|
||||
alternative composition sources, not an implicit mixture.
|
||||
- A requested debug bundle has one terminal report attempt; its persistence
|
||||
errors supplement rather than obscure the primary command error.
|
||||
- User-facing flags, paths, exit codes, and configuration fields are defined
|
||||
by their public documentation, not duplicated here.
|
||||
|
||||
## Focused Tests
|
||||
|
||||
- **internal/cli/command_contract_test.go** covers dispatch, help, syntax and
|
||||
runtime error classes, discovery, validation, and listing.
|
||||
- **internal/cli/run_contract_test.go** covers the run handoff, publication,
|
||||
debug reporting, and command-owned state collaborators.
|
||||
- **internal/cli/production_contract_test.go** covers registrar composition,
|
||||
production catalog contents, assets, and representative configuration
|
||||
validation.
|
||||
- **internal/cli/reference_contract_test.go** covers CLI reference overrides,
|
||||
origin separation, and materialization boundaries.
|
||||
- **internal/cli/state_hardening_test.go** covers safe run identity, state
|
||||
roots, and failure ordering.
|
||||
|
||||
Run **go test ./internal/cli** after changing command composition or command
|
||||
behavior. Pair it with **go test ./internal/core/config** when the configuration
|
||||
handoff changes.
|
||||
144
docs/internal/configuration.md
Normal file
144
docs/internal/configuration.md
Normal file
@@ -0,0 +1,144 @@
|
||||
# Configuration Internals
|
||||
|
||||
This document describes the maintainer-facing configuration boundary in
|
||||
**internal/core/config**. The [Configuration](../config.md) reference owns the
|
||||
file format, fields, defaults, precedence contract, and selectable keys. The
|
||||
[CLI reference](../cli.md) owns command syntax; this document does not redefine
|
||||
either interface.
|
||||
|
||||
## Boundary
|
||||
|
||||
The configuration package turns a selected YAML file and supported environment
|
||||
values into a validated, independently owned configuration. It then resolves a
|
||||
requested pipeline against a module catalog before the framework prepares or
|
||||
runs anything.
|
||||
|
||||
| Boundary | Inputs | Outputs | Does not own |
|
||||
| --- | --- | --- | --- |
|
||||
| Loading | Selected file path and environment lookup | Parsed file model and a populated **Config** | Choosing the file path or reporting a command result. |
|
||||
| Validation | **Config** | Structural configuration errors with pipeline, lane, or binding context | Module availability, capabilities, or construction. |
|
||||
| Resolution | Valid **Config**, selected pipeline and lanes, runtime reference changes, LLM override, and module catalog | **EffectiveConfig** with a **ResolvedPipeline** | Materializing reference bytes, preparing modules, execution, or filesystem state. |
|
||||
| Summary | **Config** or **EffectiveConfig** | Detached redacted payload suitable for debug summaries | Redacting arbitrary process state or provider traffic. |
|
||||
|
||||
The CLI discovers a configuration file, invokes this package, and supplies the
|
||||
result to the framework. Configuration never reads an input file, constructs a
|
||||
module, or creates output, cache, or debug paths. Those responsibilities remain
|
||||
at their respective [CLI](cli.md), [pipeline](pipeline.md), and
|
||||
[run-state](state.md) boundaries.
|
||||
|
||||
## Loading And Validation
|
||||
|
||||
The CLI loads configuration in this order:
|
||||
|
||||
1. parse the selected YAML file strictly into the file model;
|
||||
2. start from **Default**;
|
||||
3. apply the file model; and
|
||||
4. apply the supported environment overrides.
|
||||
|
||||
This establishes the public precedence order without giving environment input a
|
||||
second file schema. Loading and application reject malformed YAML, unsupported
|
||||
file versions, unknown fields, invalid values, and identifiers that are empty
|
||||
or collide after whitespace normalization. The file application also makes the
|
||||
effective extraction-worker default follow the effective LLM limit. A present
|
||||
PromptKit local-backend object requires and trims its endpoint, defaults its
|
||||
omitted concurrency limit to zero, and is copied so the parsed file model
|
||||
cannot alias the populated **Config**. A pipeline `llm_profile` is
|
||||
presence-aware: omission remains empty, while a present blank value is
|
||||
rejected and a non-empty file value is trimmed before it reaches **Config**.
|
||||
|
||||
**Config.Validate** checks configuration-only invariants before resolution. It
|
||||
rejects incompatible profile sources, invalid state-surface values, unsupported
|
||||
concurrency settings, malformed bindings and references, invalid retries, and
|
||||
invalid pipeline, step, or lane structure. PromptKit local-backend validation
|
||||
accepts only an absolute HTTP or HTTPS endpoint with a host and no user
|
||||
information, query, or fragment, and rejects a negative local concurrency
|
||||
limit. Its errors retain the closest known pipeline, lane, and binding context.
|
||||
It deliberately does not require modules to be registered: that requires a
|
||||
catalog and belongs to resolution.
|
||||
|
||||
The exact user-selectable values and validation rules are defined in
|
||||
[Configuration](../config.md). Keep additions to the file model, an
|
||||
environment override, its validation, and that reference in the same change.
|
||||
|
||||
## Effective Resolution
|
||||
|
||||
**Config.Resolve** first recomputes derived concurrency defaults and validates
|
||||
the configuration. It normalizes the requested pipeline ID, copies the selected
|
||||
profile, and passes the non-empty command-level LLM profile override, requested
|
||||
lane selection, and reference changes to the framework resolver.
|
||||
|
||||
After module and validator selection, the resolver applies the effective
|
||||
profile policy to LLM-backed bindings only: command override, binding profile,
|
||||
pipeline profile, then the prompt default. Deterministic bindings remain
|
||||
profile-free, and no second inheritance decision occurs during execution. The
|
||||
public field definitions and precedence are owned by
|
||||
[Configuration](../config.md#pipelines).
|
||||
|
||||
The framework resolver supplies defaults, selects lanes, resolves validator
|
||||
chains, checks registered module and artifact compatibility, validates module
|
||||
options, and returns the fixed ordered pipeline shape. The resulting
|
||||
**EffectiveConfig** retains the selected ID, requested selection and reference
|
||||
changes, a clone of the input configuration, and the resolved pipeline.
|
||||
Callers may therefore retain or modify their input slices and maps without
|
||||
changing the resolved result, and later consumers cannot mutate the original
|
||||
configuration through the effective value. This ownership includes the nested
|
||||
PromptKit local-backend value.
|
||||
|
||||
Resolution failures stop before module construction and source parsing. They
|
||||
include an error path for an unconfigured pipeline, missing module, missing
|
||||
capability, incompatible artifact variant, invalid option, invalid reference,
|
||||
or invalid lane selection. CLI code maps these valid-invocation failures to the
|
||||
runtime error class described in the [CLI reference](../cli.md#output-streams-and-exit-statuses).
|
||||
|
||||
## Resolved Identity And Redaction
|
||||
|
||||
The framework assigns the resolved pipeline a deterministic SHA-256 digest
|
||||
after defaults, lane selection, module bindings, reference bindings, validator
|
||||
chains, effective LLM profiles, and artifact schema identity have been
|
||||
resolved. The digest excludes
|
||||
its own stored value. It identifies resolved composition rather than raw YAML
|
||||
bytes, a debug payload, or all runtime state. The CLI records it as invocation
|
||||
provenance before execution; cache and checkpoint identity have additional
|
||||
owners in [Run State Internals](state.md).
|
||||
|
||||
Configuration summaries must use **Redacted**, **RedactedSummaryPayload**, or
|
||||
**RedactedResolvedPipelinePayload**, never a direct configuration marshal.
|
||||
Those methods copy every binding and nested option container, replace values
|
||||
whose key is credential-shaped with **[REDACTED]**, and omit materialized
|
||||
reference content while retaining safe binding and reference provenance. The
|
||||
payload must not alias the source configuration or resolved pipeline.
|
||||
PromptKit's local endpoint and concurrency limit are preserved as non-secret
|
||||
configuration metadata in the independently owned summary; the object contains
|
||||
no credential value. This redaction is deliberately narrow: it protects
|
||||
configuration summaries and does not authorize recording arbitrary environment
|
||||
values or provider requests.
|
||||
|
||||
## Invariants To Preserve
|
||||
|
||||
- Defaults, YAML values, and environment values are applied in one direction;
|
||||
later sources may override only their supported operational settings.
|
||||
- A configuration is structurally valid before it is resolved, and a resolved
|
||||
pipeline is compatible with the supplied catalog before preparation begins.
|
||||
- Whitespace-normalized identifiers are unique wherever they identify a
|
||||
pipeline, step, lane, worker, or reference slot.
|
||||
- Resolution and summary generation return detached data. Redaction must cover
|
||||
every configured and resolved binding, including nested validator bindings.
|
||||
- The resolved digest changes when resolved composition changes and never
|
||||
includes itself.
|
||||
|
||||
## Focused Tests
|
||||
|
||||
- **internal/core/config/file_config_contract_test.go** covers strict file
|
||||
parsing, normalization, file application, and structural rejection.
|
||||
- **internal/core/config/env_contract_test.go** covers supported operational
|
||||
overrides and their precedence.
|
||||
- **internal/core/config/validation_contract_test.go** covers configuration
|
||||
invariants and contextual failures.
|
||||
- **internal/core/config/effective_config_contract_test.go** covers defaults,
|
||||
selections, overrides, resolution context, digest changes, and ownership.
|
||||
- **internal/core/config/redaction_test.go** covers recursive credential
|
||||
redaction, reference-content exclusion, and non-aliasing payloads.
|
||||
|
||||
Run **go test ./internal/core/config** after changing this boundary. Changes to
|
||||
the handoff or resolved-composition semantics also need the focused framework
|
||||
pipeline tests.
|
||||
155
docs/internal/dnd.md
Normal file
155
docs/internal/dnd.md
Normal file
@@ -0,0 +1,155 @@
|
||||
# D&D Module Internals
|
||||
|
||||
This guide records the conventions shared by the production D&D module family.
|
||||
It complements [Module Internals](modules.md), which owns generic registration
|
||||
and extension mechanics, and [Configuration](../config.md), which owns the
|
||||
selectable keys, bindings, reference syntax, and default validator chains.
|
||||
|
||||
## Durable Artifact Contracts
|
||||
|
||||
The six lanes have separate durable wire contracts. This guide deliberately
|
||||
does not repeat their JSON shapes or schemas.
|
||||
|
||||
| Lane | Durable contract |
|
||||
| --- | --- |
|
||||
| Spells | [spell artifacts](../integrations/dnd-spell-artifacts.md) |
|
||||
| NPCs | [NPC artifacts](../integrations/dnd-npc-artifacts.md) |
|
||||
| Combat turns | [combat-turn artifacts](../integrations/dnd-combat-turn-artifacts.md) |
|
||||
| Item events | [item-event artifacts](../integrations/dnd-item-event-artifacts.md) |
|
||||
| NPC interactions | [NPC-interaction artifacts](../integrations/dnd-npc-interaction-artifacts.md) |
|
||||
| Scene descriptions | [scene-description artifacts](../integrations/dnd-scene-description-artifacts.md) |
|
||||
|
||||
## Family Composition
|
||||
|
||||
The D&D registrar registers the family’s artifact codecs, extractors, typed
|
||||
append-order mergers, normalizers, validators, prompt assets, fallback LLM
|
||||
profile asset, and default validator chains. Each extractor and normalizer has
|
||||
a stable module spec, explicit execution class, strict option decoding, and a
|
||||
typed builder. Scene chunking, every extractor, and NPC normalization are
|
||||
registered as `llm_backed`; the remaining current D&D mergers and normalizers
|
||||
are `deterministic`. The metadata is available to catalog inspection and
|
||||
resolved-pipeline debug data and determines which selected bindings inherit the
|
||||
pipeline profile. Configuration remains the canonical owner of the exact keys,
|
||||
profile precedence, and validator order.
|
||||
|
||||
Private structured-LLM response schemas are deliberately minimal. They reject
|
||||
invalid JSON structure, missing required fields, incompatible types, and
|
||||
unknown fields, while preserving semantic candidates for deterministic
|
||||
validation. Do not promote a private response envelope into a durable schema;
|
||||
the contracts above define durable data.
|
||||
|
||||
## Prompt Construction
|
||||
|
||||
D&D extractors assemble prompts from an ordered manifest of shared and
|
||||
module-owned assets. Reuse the shared D&D system, evidence, identity,
|
||||
reference, and transcript assets instead of copying their text into individual
|
||||
modules. A manifest’s declared sequence, including cache-control placement, is
|
||||
part of the prompt behavior.
|
||||
|
||||
Every maintained D&D LLM prompt selects `dnd-extraction` as its default
|
||||
profile. The D&D registrar embeds that fallback profile with the maintained
|
||||
OpenRouter model, timeout, and service-tier policy. An operator may provide a
|
||||
complete profile with the same ID through the configured PromptKit source; that
|
||||
definition replaces the fallback rather than merging with it. The fallback
|
||||
leaves reasoning and optional sampling controls unspecified. Deployment profile
|
||||
selection and the maintained operator example are documented in
|
||||
[Configuration](../config.md#promptkit-profiles).
|
||||
|
||||
All extraction prompts share this four-message rendered prefix: the system
|
||||
message without cache control, the identity message without cache control, the
|
||||
campaign-reference message with ephemeral cache control, and the chunk
|
||||
transcript message with ephemeral cache control. This gives equivalent
|
||||
extraction requests the same reusable prefix through their source material.
|
||||
|
||||
Extraction-evidence policy, generated NPC registries, spell catalogs, module
|
||||
tasks, and instructions follow the transcript because they are not universal
|
||||
across all extraction lanes. The final instructions message carries ephemeral
|
||||
cache control; evidence, registry, catalog, and task messages do not. Preserve
|
||||
this division when changing an extractor or its assets so prompt-cache behavior
|
||||
remains stable.
|
||||
|
||||
The other D&D LLM prompts intentionally follow different patterns. Scene
|
||||
chunking has no sibling extraction lane with which to share its full transcript,
|
||||
so it renders campaign references before its task and instructions, then places
|
||||
the cacheable full transcript last. NPC normalization keeps its task and
|
||||
cacheable instructions before the candidate collection, followed by the
|
||||
cacheable transcript windows: candidates must be available before their
|
||||
supporting evidence is evaluated, and those windows are not a cross-lane
|
||||
prefix. Mounted assets and their declared message order determine the prompt
|
||||
fingerprint, so intentional prompt edits continue to invalidate stale
|
||||
checkpoints.
|
||||
|
||||
All extractors use the shared prompt-input preparation rules. The current chunk
|
||||
is copied into transcript material; player, party, glossary, and compatible
|
||||
campaign references are context for disambiguation, not source evidence.
|
||||
Reference prompt material is canonically ordered before it is rendered, which
|
||||
keeps equivalent inputs stable across runs.
|
||||
|
||||
## Evidence, Candidates, And Normalization
|
||||
|
||||
The current transcript is the only durable evidence source. Extractors assign
|
||||
the current source identity, preserve candidate evidence ranges for validators,
|
||||
and canonically order or remove exact duplicate ranges without asking the
|
||||
model to repair semantic errors. Campaign context and generated artifacts may
|
||||
ground names or control routing, but they never establish evidence for a D&D
|
||||
result.
|
||||
|
||||
Default chains keep responsibilities separate: structural validators assess the
|
||||
candidate, source-reference validators resolve cited ranges against the current
|
||||
source, durable-schema validation checks an approved representation, and
|
||||
relatedness validators report advisory evidence concerns. The configured order
|
||||
is documented in
|
||||
[Configuration](../config.md#production-validator-keys-and-default-chains).
|
||||
|
||||
Normalizers are deterministic for spells, combat turns, item events, NPC
|
||||
interactions, and scene descriptions. They canonicalize display values and
|
||||
evidence, use source-document order for stable output, and issue bounded
|
||||
warnings for changes or collapsed duplicates. The NPC normalizer is the
|
||||
intentional exception: it first produces a deterministic candidate set, then
|
||||
uses a bounded structured-LLM proposal to reconcile identity groups. Invalid
|
||||
or unusable proposals retain the deterministic result and surface retry or
|
||||
fallback diagnostics; the model does not directly replace durable records.
|
||||
|
||||
## Generated References And Grounding
|
||||
|
||||
Normalized D&D artifacts can be handed to a later step through a generated
|
||||
reference binding. The framework verifies artifact compatibility and retains
|
||||
producer provenance; consumers resolve the handed-off artifact into an
|
||||
immutable, validated projection for each operation. External files are checked
|
||||
during preparation, while generated artifacts are resolved at the handoff.
|
||||
|
||||
NPC registries are names-only grounding projections: they may canonicalize
|
||||
actors for spells and combat turns and are required for NPC interactions, but
|
||||
they do not supply evidence. Scene-description registries are eligibility-only
|
||||
projections: they retain the current chunk’s classification data, not scene
|
||||
prose or evidence, and exist to route combat extraction.
|
||||
|
||||
## Lane-Specific Rules
|
||||
|
||||
The following differences are intentional and should remain explicit when a
|
||||
shared helper changes.
|
||||
|
||||
| Lane | Intentional behavior |
|
||||
| --- | --- |
|
||||
| Spells | May use a spell-catalog overlay and optional NPC grounding; the catalog validator supplies domain-specific semantic checks. |
|
||||
| NPCs | Does not consume an NPC registry. Its normalizer is the LLM-assisted reconciliation exception described above. |
|
||||
| Combat turns | Requires a scene-description artifact. It calls the LLM only for an exact `combat` classification; exact non-combat classifications return an accepted empty result, while missing or mismatched classifications return an empty result with a bounded warning. Optional NPC grounding never becomes evidence. |
|
||||
| Item events | Uses campaign context for disambiguation but has no NPC-registry or scene-description dependency. |
|
||||
| NPC interactions | Requires the normalized NPC registry at extraction and normalization, using it for canonical actor grounding only. |
|
||||
| Scene descriptions | Produces the classifications consumed by combat routing; it does not consume an NPC registry or provide evidence for combat artifacts. |
|
||||
|
||||
The combat and scene-description contracts describe their exact handoff and
|
||||
empty-result behavior in more detail:
|
||||
[combat turns](../integrations/dnd-combat-turn-artifacts.md) and
|
||||
[scene descriptions](../integrations/dnd-scene-description-artifacts.md).
|
||||
|
||||
## Focused Verification
|
||||
|
||||
When changing D&D behavior, test the affected codec, extractor, normalizer,
|
||||
validator, prompt-asset manifest, and registry projection. Also test generated
|
||||
handoffs at the integration boundary and run the full D&D module suite:
|
||||
|
||||
~~~sh
|
||||
go test ./internal/modules/dnd/...
|
||||
go test ./internal/modules/integration/...
|
||||
~~~
|
||||
@@ -1,222 +1,251 @@
|
||||
# LLM Runtime Internals
|
||||
|
||||
`internal/framework/llm` implements Notarius's transport boundary for structured
|
||||
completion. It contains the Scriptorium adapter, concurrency scheduler,
|
||||
prompt/schema registries, selected-profile recording, and provider-error
|
||||
redaction.
|
||||
`internal/framework/llm` is Notarius’s provider-independent structured
|
||||
completion boundary. It adapts framework requests to PromptKit, bounds
|
||||
provider calls, assembles registered prompt and schema assets, records selected
|
||||
profiles, and redacts provider errors. The architectural boundary is defined in
|
||||
[Architecture](../policy/architecture.md#llm-boundary); profile sources,
|
||||
credentials, and concurrency settings belong in
|
||||
[Configuration](../config.md#promptkit-profiles) and
|
||||
[Configuration](../config.md#concurrency-output-cache-and-debug).
|
||||
|
||||
Provider-neutral ownership rules are defined in
|
||||
[Architecture](../policy/architecture.md#llm-boundary). Profile sources,
|
||||
credentials, and concurrency settings are defined in
|
||||
[Configuration](../config.md).
|
||||
## Structured Completion Boundary
|
||||
|
||||
## Structured Contract
|
||||
Modules and LLM-backed validators depend only on
|
||||
`contracts.StructuredLLMClient`. A completion request supplies a prompt ID and
|
||||
version, optional profile and session IDs, named input material, variables, and
|
||||
a caller-owned decode target. The successful response returns the validated raw
|
||||
structured bytes together with non-secret provider, model, profile, and token
|
||||
metadata.
|
||||
|
||||
Modules and LLM-backed validators depend on
|
||||
`contracts.StructuredLLMClient.CompleteStructured`. A request identifies a
|
||||
prompt and optional profile/session, supplies named input materials and
|
||||
variables, and provides a caller-owned decoding target. A successful response
|
||||
contains the validated raw structured bytes plus non-secret provider, model,
|
||||
profile, and token metadata.
|
||||
The caller owns the domain behavior: it chooses the prompt, prepares inputs,
|
||||
selects the private response schema, and interprets the decoded result. The
|
||||
adapter does not own source evidence, artifact conversion, normalization, or
|
||||
durable schemas. Those responsibilities remain with the module and its
|
||||
[integration contract](../integrations/).
|
||||
|
||||
The caller owns prompt selection, response-schema selection, and interpretation
|
||||
of the decoded result. `LLMInputMaterial` keeps source and reference bytes with
|
||||
their origin metadata so the adapter can pass named artifacts to Scriptorium
|
||||
without exposing Scriptorium types through stage contracts.
|
||||
`PromptKitClient` validates the request target and prompt identity, maps each
|
||||
named material to a PromptKit inline artifact while preserving its origin URI,
|
||||
maps the trimmed request session to PromptKit's direct per-run session field,
|
||||
retains the same value as the `session_id` prompt variable for maintained
|
||||
prompt compatibility, and forwards profile selection. It then creates one
|
||||
frozen prepared execution, captures its caller-owned credential-redacted
|
||||
details for debug material, and executes that exact snapshot through
|
||||
PromptKit's prepared-execution boundary. The direct field
|
||||
is authoritative for provider session behavior. A session ID is a stable,
|
||||
non-secret correlation identifier and may be exposed to providers and provider
|
||||
observability. The adapter returns PromptKit’s validated raw bytes rather than
|
||||
re-encoding the decoded target. An empty optional material is represented as
|
||||
one space so its named input is retained by PromptKit.
|
||||
|
||||
## Production Construction
|
||||
Client construction may also receive a run-wide reasoning-effort override from
|
||||
the CLI factory boundary. The adapter copies the caller-owned pointer and
|
||||
creates a fresh PromptKit execution override for each request: a nil pointer
|
||||
inherits the selected profile, a non-empty value replaces it, and an empty
|
||||
value clears inherited reasoning. The CLI's mutually exclusive
|
||||
`--reasoning-effort` and `--clear-reasoning-effort` controls select those
|
||||
states. With neither flag, profile behavior remains unchanged. Because
|
||||
production constructs one shared client, the selected state applies uniformly
|
||||
to module calls, retries, and LLM-backed validators for the whole run.
|
||||
|
||||
`internal/cli` constructs the production runtime by:
|
||||
An empty request profile lets the prompt select its configured default. Before a
|
||||
run begins, the CLI asks the adapter to inspect every explicit profile on the
|
||||
resolved selected LLM-backed bindings and validators, including inherited
|
||||
pipeline profiles. Inspection resolves the profile and its selected backend and
|
||||
target without loading a prompt, reading credentials, admitting capacity, or
|
||||
contacting a provider, so a missing or invalid explicit profile fails before
|
||||
stage execution while a valid `api_key_env` may remain unset. Calls record the
|
||||
profile actually selected by PromptKit. The recorder trims and deduplicates
|
||||
non-secret profile identity, provider, model, selected backend ID, and
|
||||
effective reasoning values for manifest use. Entries that differ in backend or
|
||||
reasoning remain distinct and deterministically ordered. Endpoint-only profiles
|
||||
retain an empty backend ID, which the published JSON omits. Successful
|
||||
completion responses and recorded profile manifests identify the adapter
|
||||
provider as `promptkit`.
|
||||
|
||||
1. allocating the asset registry populated by the generic, Seriatim, and D&D
|
||||
package-family registrars;
|
||||
2. creating a `ScriptoriumClient` from the effective profile source;
|
||||
3. attaching an `LLMProfileRecorder`;
|
||||
4. creating a scheduler from the effective concurrency limit;
|
||||
5. returning a `ScheduledClient` wrapper;
|
||||
6. decorating that shared client before preparation when debug recording is
|
||||
enabled; and
|
||||
7. injecting that one shared client into complete pipeline preparation before
|
||||
the source file is read or the runner is invoked.
|
||||
The CLI's profile-inspection engine and the production adapter use the same
|
||||
profile-source construction to apply the configured profile directory or file,
|
||||
the optional registered fallback profile assets, and the optional conventional
|
||||
`local` backend. Preflight therefore resolves the same profile sources and
|
||||
backend membership as runtime without performing generation. Fallback assets
|
||||
are mounted only when at least one source is registered. The production D&D
|
||||
registrar contributes its `dnd-extraction` fallback, and the maintained D&D
|
||||
prompts select that logical ID by default. PromptKit owns source precedence and
|
||||
profile parsing: an operator-provided matching profile takes precedence over a
|
||||
fallback profile without Notarius merging either document.
|
||||
When the registration is absent, a profile selecting `backend: local` fails
|
||||
inspection instead of falling back to a built-in or endpoint-only target.
|
||||
|
||||
The D&D scene chunker and spell, NPC, and combat extractors retain this
|
||||
injected client and use it for every structured completion. Operation requests
|
||||
do not carry an LLM client.
|
||||
Before execution, the adapter also contributes a non-secret checkpoint
|
||||
fingerprint for the effective PromptKit profile source. It combines the
|
||||
identity of PromptKit's compiled-in profile catalog with a deterministic digest
|
||||
of every YAML profile in the configured profile directory, or of the configured
|
||||
profile file, and a deterministic digest of the flattened fallback profile
|
||||
assets. The fingerprint contains neither profile content nor source paths. It
|
||||
covers inherited pipeline profiles, explicit binding profiles, and
|
||||
prompt-selected defaults, so changing a model or other profile setting cannot
|
||||
reuse checkpoints created under the
|
||||
prior profile source. This cache identity is independent of durable
|
||||
profile provenance: run manifests continue to list only profiles actually
|
||||
observed during LLM calls. When the local backend is registered, a second
|
||||
fingerprint hashes its trimmed endpoint behind a stable marker. Changing that
|
||||
semantic execution target invalidates checkpoint reuse. The raw endpoint is not
|
||||
stored in checkpoint identity, and the local concurrency limit is excluded
|
||||
because it changes scheduling rather than execution semantics.
|
||||
|
||||
The CLI separately gathers explicit profile IDs from resolved LLM-capable stage
|
||||
and validator bindings. It prepares a small internal check prompt for each ID so
|
||||
missing or invalid profiles fail before pipeline execution. The runtime profile
|
||||
override syntax and scope are defined in the
|
||||
[CLI reference](../cli.md#run); binding rules are defined in
|
||||
[Configuration](../config.md#module-bindings).
|
||||
## Shared Provider-Call Limit
|
||||
|
||||
## Scriptorium Adapter
|
||||
Production construction creates one PromptKit client and wraps it in one
|
||||
scheduled client. The scheduler has a fixed, positive permit limit, serves
|
||||
queued calls in FIFO order, and removes a queued call when its context is
|
||||
cancelled. A granted permit is released exactly once on every completion path.
|
||||
|
||||
`ScriptoriumClient` converts a Notarius request into a Scriptorium `RunRequest`.
|
||||
It validates the decoding target and prompt identity, maps named input materials
|
||||
to inline artifacts, forwards explicit profile and session context, delegates
|
||||
rendering/provider execution/structured validation, and unmarshals successful
|
||||
JSON into the caller target.
|
||||
The scheduled wrapper surrounds every `CompleteStructured` call, so concurrent
|
||||
lanes, pipeline retries, and LLM-backed validators share the same provider-call
|
||||
ceiling. This ceiling is independent of pipeline worker concurrency; changing
|
||||
worker counts cannot exceed the configured LLM limit. The configuration field
|
||||
and its effective default are owned by
|
||||
[Configuration](../config.md#concurrency-output-cache-and-debug).
|
||||
|
||||
Empty optional input material is represented by a single space so Scriptorium
|
||||
retains the named input. The client returns Scriptorium's validated structured
|
||||
bytes rather than re-encoding the caller target, allowing modules to preserve
|
||||
the runtime result exactly.
|
||||
|
||||
Selected profile, provider, model, and token metadata are mapped into the
|
||||
Notarius response. The recorder deduplicates profiles by identity and supplies
|
||||
manifest-safe profile summaries after actual calls; manifest population does
|
||||
not guess the selected prompt default in advance.
|
||||
|
||||
Generated-output validation failures and provider failures are wrapped with
|
||||
prompt context. Error strings pass through bearer-token redaction before they
|
||||
cross the runtime boundary.
|
||||
|
||||
## Scheduling
|
||||
|
||||
`Scheduler` uses a bounded permit count and a FIFO waiter queue. Immediate
|
||||
acquisition increments the in-flight count; queued acquisition waits for a
|
||||
permit or context cancellation. Cancellation removes a queued waiter, while a
|
||||
cancelled waiter that has already received a permit releases it.
|
||||
|
||||
`ScheduledClient` acquires a permit around each structured completion and
|
||||
defers release on every result path. The effective limit and default are
|
||||
configuration facts in [Configuration](../config.md#defaults).
|
||||
|
||||
This provider-call ceiling is independent of the pipeline's extract worker
|
||||
limit. Concurrent lanes, retries, and validators all use the same scheduled
|
||||
client, so increasing framework workers cannot exceed `total_llm`. Pipeline
|
||||
dispatch and cancellation mechanics are documented in
|
||||
[Pipeline Internals](pipeline.md#execution-flow).
|
||||
PromptKit applies a second, independent admission limit when the selected
|
||||
profile names a limited backend. It sits beneath the Notarius scheduled client,
|
||||
so it may narrow but cannot expand the application-wide limit. Built-in
|
||||
OpenRouter profiles select PromptKit's reserved backend and its upstream
|
||||
capacity policy. A positive configured local-backend limit bounds active local
|
||||
generations inside PromptKit; zero leaves that backend unlimited there.
|
||||
Endpoint-only profiles do not select a PromptKit backend and remain limited
|
||||
only by the Notarius scheduler.
|
||||
|
||||
## Prompt And Schema Assets
|
||||
|
||||
`AssetRegistry` combines caller-owned prompt filesystems under stable prefixes
|
||||
and rejects invalid or conflicting registrations. Production module packages
|
||||
register their own prompt and schema assets; generic framework code contains no
|
||||
D&D prompt content. `internal/framework/promptfs` provides the domain-neutral
|
||||
filesystem composition helper used to combine module-owned files with shared
|
||||
domain prompt fragments.
|
||||
An `AssetRegistry` collects prompt, schema, and optional fallback-profile
|
||||
filesystems from production module families. It flattens registered roots into
|
||||
the corresponding PromptKit filesystems and rejects invalid roots, unreadable
|
||||
assets, duplicate paths, and missing prompt or schema files during preparation.
|
||||
Fallback assets receive a safe content digest for checkpoint identity; raw
|
||||
paths and bytes are never included. The framework’s `promptfs` helper combines
|
||||
module-owned prompt files with reusable domain fragments without making the
|
||||
framework depend on D&D content.
|
||||
|
||||
The D&D scene chunker and spell, NPC, and combat-turn extractors each declare an
|
||||
ordered prompt asset manifest. The manifest lists the package-owned YAML and
|
||||
Markdown files, then the exact shared fragments rendered by that prompt; the
|
||||
same ordered list drives both filesystem mounting and the prompt fingerprint.
|
||||
Unused shared assets are neither mounted nor fingerprinted. Universal
|
||||
extraction-evidence and output policy lives only in the shared extraction
|
||||
assets; package-owned prompt files retain artifact-specific rules. The scene
|
||||
prompt keeps its separate output rule because it does not render the
|
||||
extraction-evidence asset.
|
||||
Each LLM-backed module owns its prompt declaration, package-specific assets,
|
||||
and private response schema. Shared D&D wording is owned by the D&D shared
|
||||
asset package; the detailed D&D conventions are in
|
||||
[D&D Module Internals](dnd.md). The mounted prompt assets used by a module also
|
||||
determine its prompt fingerprint. Schema loaders validate JSON, attach identity
|
||||
and digest metadata, make defensive copies, and expose diagnostics without raw
|
||||
schema bytes.
|
||||
|
||||
### D&D Extraction Prompt Ordering And Cache Boundaries
|
||||
Private response schemas validate a model transport envelope. They are not the
|
||||
durable artifact schema and should not be documented as an external wire
|
||||
contract. Durable formats and compatibility rules remain in the
|
||||
[integration contracts](../integrations/).
|
||||
|
||||
D&D extraction prompts order messages from the most reusable content to the
|
||||
most variable content. New extraction lanes use these tiers in order:
|
||||
## Prompt Maintenance And Backend Caching
|
||||
|
||||
1. universal shared content, including the system, extraction-evidence, and
|
||||
in-world identity messages;
|
||||
2. stable campaign or run context shared across lanes, including campaign
|
||||
references;
|
||||
3. stable subset- and lane-specific context and instructions, including an NPC
|
||||
registry, catalog, task, or extraction instructions when applicable;
|
||||
4. the chunk transcript as the final user message.
|
||||
Prompt message order and shared asset bytes are runtime behavior. Backend cache
|
||||
reuse depends on identical preceding roles, rendered bytes, and cache-control
|
||||
metadata—not merely equivalent meaning. Keep reusable shared assets
|
||||
byte-identical and preserve each prompt’s declared ordering and cache controls
|
||||
when editing it.
|
||||
|
||||
This ordering lets requests reuse the longest identical prefix before the
|
||||
per-chunk transcript changes. Cache reuse requires the preceding message
|
||||
sequence and content to be exactly identical; semantic similarity is not
|
||||
sufficient. Cache boundaries belong at the ends of reusable stable tiers,
|
||||
subject to the provider's cache-boundary limit. The shared identity and
|
||||
campaign-reference messages form the first two extraction boundaries. Spell
|
||||
and combat prompts add a boundary at the shared NPC registry. Each extraction
|
||||
prompt places its final boundary on its lane-specific instructions, immediately
|
||||
before the transcript. The transcript does not carry cache control because no
|
||||
reusable content follows it.
|
||||
For sibling prompts that can reuse the same source material, order universal
|
||||
shared context first, request source material next, and module-specific
|
||||
suffixes last. Put a cache boundary at a reusable prefix that is useful to the
|
||||
backend. Redundant intermediate cache boundaries do not extend that reusable
|
||||
prefix and add no value.
|
||||
|
||||
Accordingly, the common prefix of all three extraction prompts is system,
|
||||
extraction evidence, identity, and campaign references. The NPC prompt then
|
||||
renders task, instructions, and transcript. Spell renders immediate resolution,
|
||||
NPC registry, catalog, task, instructions, and transcript. Combat renders
|
||||
immediate resolution, NPC registry, task, instructions, and transcript. The
|
||||
scene chunker is not an extraction lane: it retains its separate system,
|
||||
transcript, campaign-reference, task, and instruction order and marks its
|
||||
transcript and campaign-reference messages ephemeral.
|
||||
Prompt-family owners may choose a different sequence when their inputs and
|
||||
reuse pattern differ. The D&D family’s extraction, scene-chunking, and NPC
|
||||
normalization policies are maintained in [D&D Module Internals](dnd.md#prompt-construction).
|
||||
Do not add tests that enforce prompt prose; prompt tests should verify the
|
||||
meaningful input placement and cache controls of the prompt being changed.
|
||||
|
||||
Shared wording belongs in the canonical assets under
|
||||
`internal/modules/dnd/shared`; extraction packages reference those assets in
|
||||
their manifests instead of copying similar text into package-local files.
|
||||
Package-local assets contain only lane-specific content. An extraction lane may
|
||||
depart from the tier order only when prompt-quality evidence or a provider
|
||||
constraint makes the exception necessary; document the exception and rationale
|
||||
here when it becomes implemented behavior.
|
||||
## Validation, Repair, And Retries
|
||||
|
||||
Schema helpers load embedded JSON Schema with identity and digest metadata,
|
||||
return defensive copies, and expose a diagnostics map that omits schema bytes.
|
||||
The small framework registry contains only generic test schemas; production
|
||||
schemas remain package-owned.
|
||||
PromptKit performs prompt rendering, provider execution, and the prompt’s
|
||||
structured-output validation. The adapter reports an empty result, validation
|
||||
failure, empty structured body, or decode failure as
|
||||
`ErrInvalidStructuredOutput`, while retaining the returned raw bytes and debug
|
||||
material when they exist. Provider failures remain operational errors rather
|
||||
than output-validation failures.
|
||||
|
||||
The spell, NPC, and combat extractors' package-owned prompts declare their
|
||||
structured JSON inputs and private response schemas. Each private response
|
||||
schema remains separate from its durable artifact codec schema; this work does
|
||||
not use shared schema fragments or schema generation. The combat private schema
|
||||
owns the transport envelope—required fields, JSON types, nullability, and
|
||||
unknown-field rejection—while its deterministic validators own semantic
|
||||
constraints such as enum membership, non-empty values and collections, and
|
||||
positive numbers. The spell extractor's prompt declares a required
|
||||
`application/json` `spell_catalog` input and an optional `application/json`
|
||||
`npcs` input. The extractor generates
|
||||
the catalog input from its prepared
|
||||
effective catalog as `{"spell_names":[...]}` using sorted canonical names only.
|
||||
The shared D&D prompt assets include a generic NPC grounding fragment directly
|
||||
after the campaign reference message for spell and combat prompts. When an NPC
|
||||
registry is bound, the
|
||||
domain registry boundary strictly decodes and identity-validates one durable
|
||||
artifact, re-encodes canonical JSON, and generates a semantic digest over
|
||||
those bytes. The unbound input is exactly `{"npcs":[]}`. Input digests cover
|
||||
the generated bytes; manifests record catalog identity and optional NPC
|
||||
registry digest/count rather than names, aliases, overlay bytes, registry
|
||||
paths, or source metadata. Combat prompt, response-schema, mapping,
|
||||
normalization, identity, and bound-registry fingerprints remain separate
|
||||
semantic inputs to checkpoint identity.
|
||||
When PromptKit rejects backend admission before generation, the adapter maps
|
||||
`promptkit.ErrCapacityExceeded` to
|
||||
`contracts.ErrLLMCapacityExceeded`, retaining prompt context and a redacted
|
||||
upstream diagnostic without exposing the PromptKit sentinel or capacity-error
|
||||
type as a framework contract. When supplied, the normalized selected backend
|
||||
ID appears only in that safe application-owned diagnostic context. A canceled
|
||||
caller context takes precedence. The adapter does not retry capacity failures;
|
||||
the pipeline's existing binding attempt policy sees the operational error and
|
||||
decides whether to rerun the complete operation.
|
||||
|
||||
## Debug And Redaction Boundaries
|
||||
Prompt-declared repair is executed within PromptKit’s structured-output flow.
|
||||
The current production D&D prompt manifests set repair attempts to zero. That
|
||||
setting does not replace pipeline retry behavior: a binding’s configured retry
|
||||
count reruns its stage attempt after an error or rejection, and an exhausted
|
||||
rejection is a recorded output rather than a provider error. The pipeline owns
|
||||
attempt lifecycle, validation chains, and retry diagnostics; see
|
||||
[Pipeline Internals](pipeline.md#validation-retries-and-output) and the
|
||||
[binding reference](../config.md#module-bindings-and-validators).
|
||||
|
||||
The pipeline may wrap the client with a debug recorder that captures prepared
|
||||
prompt/response material for an explicitly requested debug run. Debug summaries
|
||||
and manifests receive identities, hashes, usage, and selected profile summaries
|
||||
rather than prompt, source, reference, schema, or response content.
|
||||
## Timeout Ownership
|
||||
|
||||
The Scriptorium error wrapper removes bearer credential values from surfaced
|
||||
provider errors; `RedactSecrets` and `ErrorWithSecretsRedacted` support known
|
||||
secret values elsewhere in the runtime. Config summaries use a separate
|
||||
clone-and-redact path in `internal/core/config`. These mechanisms implement the
|
||||
security invariant in
|
||||
[Architecture](../policy/architecture.md#state-output-and-safety); operator
|
||||
handling of debug data is defined in [Operations](../operations.md#debug).
|
||||
The caller context remains the outer cancellation authority. PromptKit applies
|
||||
a positive effective generation timeout as an inner request deadline; an
|
||||
explicit zero disables only that generation deadline. The HTTP client timeout
|
||||
is a separate transport-wide cap. Notarius forwards the caller context and
|
||||
does not install another timeout wrapper around PromptKit.
|
||||
|
||||
## Failure Behavior
|
||||
The selected PromptKit profile owns generation settings. Notarius binding
|
||||
retries remain outside the adapter and repeat the complete module operation
|
||||
and validation chain. PromptKit does not add a provider retry loop.
|
||||
Operator-facing behavior is summarized in
|
||||
[Operations](../operations.md#operational-limits), and the pinned upstream
|
||||
contract is identified in
|
||||
[PromptKit Integration](../integrations/pkg-promptkit.md).
|
||||
|
||||
- Invalid targets, missing prompt IDs, malformed structured output, and
|
||||
Scriptorium failures return contextual errors to the calling module.
|
||||
- Scheduler construction rejects non-positive limits; acquisition respects
|
||||
context cancellation.
|
||||
- Asset registration rejects invalid roots, missing content, and path conflicts.
|
||||
- Schema loading distinguishes missing assets, invalid JSON, and invalid
|
||||
metadata.
|
||||
- Profile validation errors occur during CLI preparation when an explicit
|
||||
selected ID cannot be prepared.
|
||||
## Observability And Redaction
|
||||
|
||||
## Tests To Inspect
|
||||
When debug recording is enabled, the pipeline decorates the shared client. The
|
||||
wrapper records prepared prompt and response material, timing, selected profile
|
||||
and backend, effective model parameters, and call identifiers in the run’s
|
||||
debug bundle, including material available from a failed structured completion.
|
||||
Effective parameters use PromptKit's stable lower-case JSON field names and may
|
||||
include `backend_id`. For a successful completion, a debug-write failure is
|
||||
surfaced; when the completion already failed, its call error remains the
|
||||
result. Debug-bundle location, retention, and handling are operational concerns
|
||||
documented in [Operations](../operations.md#debug-bundles).
|
||||
|
||||
- `internal/framework/llm/scriptorium_client_test.go`: adapter mapping and local
|
||||
HTTP integration.
|
||||
- `internal/framework/llm/scheduler_test.go` and
|
||||
`scheduled_client_test.go`: permits, FIFO behavior, cancellation, and wrapper
|
||||
release.
|
||||
- `internal/framework/llm/asset_registry_test.go` and
|
||||
`schema_registry_test.go`: asset composition, validation, and defensive
|
||||
copies.
|
||||
- `internal/framework/llm/secrets_test.go`: provider-error redaction.
|
||||
- `internal/cli/run_contract_test.go`: profile validation, production client
|
||||
wiring, manifest recording, and debug integration.
|
||||
- Module-local `scriptorium_assets_test.go` files: prompt inputs and package
|
||||
asset registration.
|
||||
Run manifests receive selected profile summaries, including optional effective
|
||||
backend and reasoning provenance, and component identities—not prompt, schema,
|
||||
source, reference, or response content. The published field semantics belong
|
||||
to the [JSON output contract](../integrations/json-output.md#manifestjson).
|
||||
Provider error text is wrapped with prompt context and bearer credentials are
|
||||
redacted before it crosses the runtime boundary. Known-secret redaction is
|
||||
available to other runtime collaborators; it does not make prompt or response
|
||||
contents safe for general logging.
|
||||
|
||||
## Failure Boundaries
|
||||
|
||||
- Construction fails for missing asset registries, mutually exclusive profile
|
||||
sources, invalid asset registration, or a non-positive scheduler limit.
|
||||
- Preparation failures, unavailable explicit profiles, provider failures, and
|
||||
context cancellation propagate to the calling stage with context.
|
||||
- Backend admission exhaustion is a provider-neutral operational error and is
|
||||
not classified as invalid structured output or validator rejection.
|
||||
- Malformed or schema-invalid provider output is classified separately as
|
||||
invalid structured output so the module or pipeline can apply its own retry
|
||||
and rejection policy.
|
||||
- Domain semantic checks, evidence decisions, and deterministic normalization
|
||||
run outside the provider adapter.
|
||||
|
||||
## Focused Verification
|
||||
|
||||
Read the LLM adapter, scheduler, asset registry, schema loader, and redaction
|
||||
tests when changing this boundary. Prompt changes also require the owning
|
||||
module’s asset tests, and retry or debug changes require focused pipeline or
|
||||
CLI coverage. The focused runtime and D&D checks are:
|
||||
|
||||
~~~sh
|
||||
go test ./internal/framework/llm/... ./internal/modules/dnd/...
|
||||
~~~
|
||||
|
||||
@@ -1,481 +1,113 @@
|
||||
# Module And Validator Internals
|
||||
|
||||
Production module and validator implementations live under their domain-first
|
||||
trees in `internal/modules`.
|
||||
The selectable keys, configuration options, reference slots, and default
|
||||
validator chain are canonical in the
|
||||
[module](../config.md#implemented-production-modules) and
|
||||
[validator](../config.md#implemented-production-validators) catalogs in
|
||||
Configuration.
|
||||
|
||||
## Extension Pattern
|
||||
|
||||
A stage module package provides a stable key, constructor, contract
|
||||
implementation, `ModuleSpec`, `Register`, and focused behavior and registration
|
||||
tests. A validator package follows the same pattern with `ValidatorSpec` and the
|
||||
validator registry. Package-family registrars compose those leaf registrations
|
||||
into the production catalog and own family-level policy such as default
|
||||
validator chains and prompt asset collection.
|
||||
|
||||
Production input, chunk, output, and D&D spell- and combat-extract packages
|
||||
register strict option decoders and run-local builders. Preparation decodes their options into
|
||||
implementation-owned values and injects dependencies plus the materialized
|
||||
reference set for the selected target. Each builder receives an isolated clone
|
||||
of that set; input and output builders receive no references. The spell and
|
||||
combat extractors are typed over the canonical D&D model. D&D validators, merge,
|
||||
and normalize use typed variants; JSON representation validators use serialized
|
||||
requests; and unconditional validators expose separate chunk and typed
|
||||
variants. The D&D production registrar registers the canonical typed spell,
|
||||
NPC, and combat implementations, including their kind-specific merge and
|
||||
normalize behavior.
|
||||
|
||||
Prepared extractors, extract validators, and codecs may be reused concurrently
|
||||
by the run-wide extract pool. Production implementations are immutable after
|
||||
construction: they retain only typed options, immutable assets, or the shared
|
||||
concurrency-safe LLM client. Implementations that introduce mutable state must
|
||||
synchronize that state without creating a separate provider scheduler.
|
||||
|
||||
Specs expose capability and execution metadata without constructing an
|
||||
implementation. Registry entries separately expose option validation and
|
||||
run-local construction. Chunk, extract, merge, and normalize modules that accept
|
||||
auxiliary material declare identical reference slots from both
|
||||
`ReferenceSlots()` and `ModuleSpec().ReferenceSlots`; registration tests enforce
|
||||
that agreement. Runtime delivery uses the corresponding stage request's
|
||||
`References` field.
|
||||
|
||||
LLM-backed extensions own their prompt definitions and response schemas under
|
||||
package-local embedded assets. Shared filesystem composition belongs in
|
||||
`internal/framework/promptfs`; reusable D&D prompt fragments, reference
|
||||
declarations, prompt-input assembly, and source-unit/citation helpers belong in
|
||||
`internal/modules/dnd/shared`, which also owns bounded D&D diagnostics. The
|
||||
D&D scene chunker and spell, NPC, and combat-turn extractors use ordered
|
||||
package-local prompt manifests for both rendering and prompt fingerprinting, so
|
||||
only the shared fragments each prompt actually renders participate in either
|
||||
operation. Extraction prompts place stable shared and lane-specific context
|
||||
before the variable transcript and use shared assets for wording common across
|
||||
lanes. The canonical ordering and cache-boundary policy is documented in
|
||||
[LLM Runtime](llm.md#dd-extraction-prompt-ordering-and-cache-boundaries). Stage
|
||||
contracts expose only Notarius structured-completion types, not Scriptorium
|
||||
public types.
|
||||
|
||||
The shared `ChunkPromptMaterial` helper owns common transcript material
|
||||
preparation for the spell, NPC, and combat-turn extractors. It clones supplied
|
||||
source metadata, falls back to the materialized chunk when content is absent,
|
||||
checks that content remains chunk-identical, and fills only the common default
|
||||
fields. Extractors retain their request validation and wrap helper errors with
|
||||
their module context.
|
||||
|
||||
Reference material may inform a module or prompt but must not become source
|
||||
evidence. The resolver and materializer behavior is described in
|
||||
[Pipeline Internals](pipeline.md#reference-materialization).
|
||||
|
||||
## Domain Reference Data
|
||||
|
||||
### `internal/modules/dnd/spells/catalog`
|
||||
|
||||
The spell catalog package owns the embedded, versioned D&D 5e 2014 SRD spell
|
||||
reference data. Its strict JSON asset contains one canonical record per spell,
|
||||
including spell level and all applicable class memberships. `LoadSRD5E2014`
|
||||
validates catalog identity, provenance metadata, ordering, uniqueness, levels,
|
||||
classes, aliases, and lookup-key collisions before exposing immutable copies.
|
||||
|
||||
Lookup is case-insensitive and normalizes whitespace and common apostrophe
|
||||
variants while preserving source punctuation in canonical display names. The
|
||||
catalog contains 319 unique spells and 779 class memberships. Source and
|
||||
license details live beside the asset in `SOURCES.md`. This domain-owned data is
|
||||
separate from `internal/modules/dnd/shared`, which is reserved for reusable
|
||||
prompt and source-reference machinery.
|
||||
|
||||
`ResolveEffectiveCatalog` builds the immutable recognition view used by the
|
||||
spell extractor and catalog validator. It starts with the embedded SRD catalog
|
||||
and optionally applies one strict JSON overlay from the `spell_catalog` item in
|
||||
a materialized reference set. Overlay catalogs are ordered by ID, may add names
|
||||
and aliases, and may augment an existing canonical spell without replacing its
|
||||
display name. Cross-spell lookup collisions are errors. The effective view
|
||||
exposes sorted canonical names, normalized lookup, overlay identities, and a
|
||||
semantic digest; overlay content remains contextual reference material rather
|
||||
than source evidence. Its external JSON contract is defined in the
|
||||
[spell-catalog overlay contract](../integrations/dnd-spell-catalog-overlays.md).
|
||||
|
||||
### `internal/modules/dnd/npcs/identity`, `internal/modules/dnd/npcs/registry`, and `internal/modules/dnd/codec/npcs`
|
||||
|
||||
The NPC identity package owns Unicode comparison keys, deterministic
|
||||
`npc:sha256:` IDs, display normalization, and whole-registry collision issues.
|
||||
The registry package resolves one optional normalized artifact through the
|
||||
strict codec, validates whole-registry identity, canonicalizes its JSON, and
|
||||
provides immutable records, prompt input, semantic digest, count, and exact
|
||||
canonical-name/alias lookup. External files cross this boundary during
|
||||
preparation; generated artifacts cross it at the ordered step handoff. It owns
|
||||
the `npcs` slot and its bounded, content-safe validation failures. NPC source
|
||||
references are durable provenance and are not treated as evidence for a
|
||||
consuming pipeline. The NPC codec owns the strict durable `dnd/npc-list` JSON
|
||||
boundary and exposes candidate versus approved encode/decode operations.
|
||||
|
||||
The `internal/modules/dnd/codec/combatturns` package owns the durable
|
||||
`dnd/combat-turn-list` schema and candidate versus approved JSON boundary. It
|
||||
is registered by the production D&D family registrar for the selectable combat
|
||||
lane.
|
||||
|
||||
## Input Adapter
|
||||
|
||||
### `internal/modules/seriatim/input/transcript`
|
||||
|
||||
The adapter decodes the supported transcript JSON, selects the source identity,
|
||||
computes canonical source provenance, validates segments, and maps each segment
|
||||
into a generic source unit with a self-reference plus speaker and timestamp
|
||||
metadata. It accepts no module options. Its spec advertises the transcript
|
||||
capabilities consumed by D&D modules.
|
||||
|
||||
Parsing is strict about required values and duplicate unit IDs but deliberately
|
||||
ignores unrelated Seriatim fields. The external format and derived-identity
|
||||
rules are defined in the
|
||||
[Seriatim contract](../integrations/seriatim.md).
|
||||
|
||||
## Chunkers
|
||||
|
||||
Chunkers implement `contracts.Chunker.Plan`. A plan identifies ordered source
|
||||
unit ranges and may carry optional namespaced JSON annotations; it does not
|
||||
contain materialized chunk content. The framework canonicalizes annotations,
|
||||
validates ranges against the current source, and materializes chunk IDs,
|
||||
indexes, references, content, units, and generic metadata. Materialized source
|
||||
unit metadata is independently owned. Annotation
|
||||
namespaces remain optional data: generic framework code and downstream modules
|
||||
must not require D&D scene annotations or import `dnd/scenes`.
|
||||
|
||||
### `internal/modules/generic/chunk/units`
|
||||
|
||||
The generic chunker validates the source document and returns ranges over units
|
||||
in configured windows. Overlap changes the next window start but never reorders
|
||||
units. Framework materialization derives the resulting chunk identity and
|
||||
generic metadata from those ranges.
|
||||
|
||||
The accepted options and defaults are defined in
|
||||
[Configuration](../config.md#implemented-production-modules). Generic
|
||||
framework validation canonicalizes the returned unit slices before extraction.
|
||||
The chunker decodes its options during construction and retains only the typed
|
||||
window settings used by `Plan`.
|
||||
|
||||
### `internal/modules/dnd/chunk/scenes`
|
||||
|
||||
The scene chunker prepares a structured Scriptorium request from the full
|
||||
transcript, session, and optional D&D reference inputs. It validates the model's
|
||||
scene boundaries against source-unit IDs and converts them into deterministic
|
||||
plan ranges with optional scene annotations. Preparation injects the shared
|
||||
structured LLM client into the chunker; `Plan`
|
||||
supplies only the run-specific profile, session, source, references, and
|
||||
metadata.
|
||||
|
||||
Scene validation requires sequential, contiguous, non-overlapping coverage from
|
||||
the first source unit through the last. Scene descriptions, boundaries,
|
||||
confidence, and participants are module-owned annotations. Boundary caveats
|
||||
become warnings. Malformed
|
||||
structured output is returned as an error; there is no fallback chunker.
|
||||
|
||||
The package embeds its prompt and response schema and reports their non-secret
|
||||
identity and hashes through singleton module metadata. Shared D&D assets supply
|
||||
reference declarations and prompt inputs; their user-facing keys and accepted
|
||||
file types remain canonical in [Configuration](../config.md).
|
||||
|
||||
## Extractor
|
||||
|
||||
### `internal/modules/dnd/extract/spells`
|
||||
|
||||
The spell extractor prepares a structured request from one chunk, the
|
||||
chunk-scoped source input, the session, and optional D&D reference inputs. It
|
||||
decodes the model response, assigns the generic source identity to every source
|
||||
reference, canonicalizes duplicate references, orders spell casts by their
|
||||
earliest cited unit, and returns `dnd.SpellList`.
|
||||
|
||||
The extractor owns its private model-response DTO, embedded prompt, LLM response
|
||||
schema, strict option decoder, injected shared LLM client, and prompt/schema
|
||||
manifest metadata. During preparation it resolves the optional `spell_catalog`
|
||||
reference into an immutable effective catalog and adds a generated
|
||||
canonical-name-only JSON input to every structured completion request. Overlay
|
||||
failures therefore stop construction before source parsing or an LLM call;
|
||||
campaign references remain separate disambiguation inputs and never become
|
||||
source evidence.
|
||||
|
||||
The prompt limits each cast to its declaration and immediate resolution; it
|
||||
does not follow summoned creatures, persistent effects, or other downstream
|
||||
consequences through the scene. Shared extraction-evidence and identity rules
|
||||
require transcript-supported factual claims and the most specific in-world
|
||||
caster identity, while campaign references only disambiguate source text.
|
||||
Effects describe the session as played: model rules knowledge cannot supplement
|
||||
or correct the transcript, and nonstandard adjudication is attributed to the GM
|
||||
or table rather than stated as a universal rule. Structural source validation
|
||||
remains deterministic; semantic claim completeness is enforced through
|
||||
extraction policy and evaluation.
|
||||
|
||||
Both the extractor and deterministic catalog validator expose
|
||||
the effective base-plus-overlay semantic digest as scoped prepared-component
|
||||
checkpoint identity. Raw overlay provenance independently covers file-byte
|
||||
changes, while the semantic digest also invalidates reuse when the embedded
|
||||
catalog or catalog composition changes. The extractor additionally fingerprints
|
||||
its complete prompt assets and private response schema, so either semantic
|
||||
contract changing invalidates previously recorded extraction checkpoints. The separate
|
||||
`internal/modules/dnd/codec/spells` package
|
||||
owns the durable schema and stable JSON representation for artifact kind
|
||||
`dnd/spell-list`. The runner keeps the result typed through validators and later
|
||||
stages, using the codec only for checkpoint, debug, and output boundaries.
|
||||
Shared D&D helpers keep prompt input
|
||||
names and source-unit reference conversion consistent with the scene chunker.
|
||||
|
||||
The extractor also declares the optional `npcs` registry slot and consumes the
|
||||
immutable registry boundary from `internal/modules/dnd/npcs/registry`. An
|
||||
external registry is prepared before execution; a generated registry is
|
||||
validated and supplied at operation time. External bindings may add only
|
||||
`npc_registry_digest` and `npc_count` to module metadata and an
|
||||
`npc_registry` checkpoint fingerprint. Generated bindings are represented by
|
||||
framework handoff provenance and dependency fingerprints. The unbound prompt
|
||||
input is exactly `{"npcs":[]}` and has no registry provenance or fingerprint.
|
||||
The shared NPC grounding fragment is placed immediately after the common
|
||||
campaign reference message and is included in the spell prompt fingerprint.
|
||||
|
||||
The durable payload and manifest metadata shapes are defined in the
|
||||
[D&D spell artifact contract](../integrations/dnd-spell-artifacts.md).
|
||||
|
||||
### `internal/modules/dnd/extract/npcs`
|
||||
|
||||
The NPC extractor maps private model output to the canonical `dnd.NPCList`,
|
||||
assigns source identity and deterministic NPC IDs, and preserves source
|
||||
references for deterministic validation. It uses the shared campaign
|
||||
references only for disambiguation and does not consume the optional NPC
|
||||
registry slot. Its prompt and private response schema are package-owned. The
|
||||
prompt follows the shared D&D extraction ordering and cache policy documented
|
||||
in [LLM Runtime](llm.md#dd-extraction-prompt-ordering-and-cache-boundaries).
|
||||
|
||||
### `internal/modules/dnd/extract/combatturns`
|
||||
|
||||
The combat extractor prepares one structured request per supplied chunk using
|
||||
the shared extraction-evidence, identity, campaign-reference,
|
||||
immediate-resolution, NPC-grounding, and transcript prompt inputs. It
|
||||
maps the private response to `dnd.CombatTurnList`, assigns the current source
|
||||
identity, removes exact duplicate source ranges, and orders turns by valid
|
||||
source-document position while preserving malformed candidate fields for
|
||||
deterministic validators. Its package-owned private response schema enforces
|
||||
only the structural JSON envelope; semantic artifact constraints remain with
|
||||
the validator chain. Its prepared metadata and checkpoint fingerprints contain
|
||||
only prompt/schema/mapping identities plus an optional NPC registry digest.
|
||||
The prompt follows the shared D&D extraction ordering and cache policy
|
||||
documented in
|
||||
[LLM Runtime](llm.md#dd-extraction-prompt-ordering-and-cache-boundaries). The
|
||||
package exposes typed registration and is included in the production D&D
|
||||
registrar with the default combat extraction chain.
|
||||
|
||||
The combat normalizer accepts only the optional structured NPC registry.
|
||||
Campaign references remain extractor-only LLM context and are not materialized
|
||||
for deterministic normalization.
|
||||
|
||||
### `internal/modules/dnd/normalize/npcs`
|
||||
|
||||
The NPC normalizer performs deterministic identity-aware consolidation in
|
||||
merged input order. It unions only canonical identity or canonical/alias
|
||||
matches, retains the first display record, unions exact relationships and
|
||||
source references, rewrites unambiguous relationship targets, and leaves
|
||||
ambiguous collisions for identity validation. It exposes the identity policy
|
||||
as its local checkpoint fingerprint and emits bounded normalization warnings.
|
||||
|
||||
## Merger And Normalizer
|
||||
|
||||
### `internal/modules/generic/merge/appendorder`
|
||||
|
||||
The merger passes typed values to an injected combine function in framework
|
||||
source-chunk order. The D&D registrar specializes it with a spell-list append
|
||||
function.
|
||||
|
||||
### `internal/modules/generic/normalize/noop`
|
||||
|
||||
The normalizer returns the merged domain value unchanged and is reusable for
|
||||
any registered artifact type.
|
||||
|
||||
### `internal/modules/dnd/normalize/spells`
|
||||
|
||||
The typed spell normalizer resolves the optional `spell_catalog` reference into
|
||||
the same immutable SRD-plus-overlay effective catalog used by spell extraction
|
||||
and catalog validation. It performs no LLM calls. For each spell cast it
|
||||
canonicalizes recognized names using the catalog's case, whitespace,
|
||||
apostrophe, and alias rules; sorts source references by source identity and
|
||||
unit boundaries; removes only exact reference duplicates; and emits bounded,
|
||||
scoped warnings for each mutation or unresolved name.
|
||||
|
||||
After those per-cast changes, it collapses only casts with the same canonical
|
||||
spell, case-folded and whitespace-normalized caster, and complete non-empty
|
||||
valid source-reference set. It retains the first occurrence and its caster,
|
||||
effect, narrative description, and stable order. Unknown names, empty or
|
||||
invalid evidence, and adjacent or overlapping but different ranges remain
|
||||
unchanged for validation.
|
||||
|
||||
The normalizer exposes the effective catalog digest as its independently scoped
|
||||
`effective_catalog` checkpoint fingerprint and reports catalog base ID, digest,
|
||||
and overlay IDs as manifest metadata. Catalog contents, reference paths, and
|
||||
raw overlay bytes are not included in either surface. The normalize-stage
|
||||
reference is stage-local, so an overlay-capable pipeline binds the catalog
|
||||
independently for extraction and normalization.
|
||||
|
||||
### `internal/modules/dnd/normalize/combatturns`
|
||||
|
||||
The combat normalizer prepares an external NPC registry before execution or
|
||||
receives a generated registry at the ordered step handoff, then uses the
|
||||
immutable view during runtime. It display-normalizes combat fields,
|
||||
rewrites exact canonical-name or alias matches for actors and targets, orders
|
||||
and deduplicates source references, stable-sorts records by source-document
|
||||
position, and collapses only exact duplicate identities with fully valid
|
||||
evidence. It deep-clones output storage and emits bounded warnings scoped to
|
||||
merged input indexes. Its metadata and fingerprints identify the normalization
|
||||
and NPC identity policies. External bindings may contribute registry
|
||||
digest/count metadata; generated identity is retained in framework provenance
|
||||
and dependency fingerprints. The normalizer is included in the production D&D
|
||||
registrar with the default combat normalization chain.
|
||||
|
||||
## Output Encoder
|
||||
|
||||
### `internal/modules/generic/output/json`
|
||||
|
||||
The JSON encoder sorts normalized results by lane, derives collision-checked
|
||||
safe logical names, pretty-prints JSON payloads, and assembles the logical index,
|
||||
manifest, rejected-result, warning, and lane files. Invalid JSON, unsupported
|
||||
media types, unsafe names, and sanitized-name collisions are errors.
|
||||
|
||||
The encoder returns logical files only. The CLI places them on disk, and the
|
||||
[JSON output contract](../integrations/json-output.md) defines their external
|
||||
paths and schemas.
|
||||
|
||||
## Generic Validators
|
||||
|
||||
The generic validator implementations live under
|
||||
`internal/modules/generic/validate`.
|
||||
|
||||
The unconditional accept and reject validators provide explicit chunk and
|
||||
typed-artifact variants used primarily for controlled composition and tests.
|
||||
|
||||
The serialized JSON syntax validator uses `encoding/json` to reject malformed
|
||||
representation bytes. The serialized JSON Schema validator requires schema
|
||||
bytes, parses the instance and schema with `jsonschema`, and distinguishes
|
||||
payload rejection from schema loading or compilation errors. The framework
|
||||
serialized-validation request carries either canonical chunk bytes or artifact
|
||||
codec bytes according to its target context. Neither validator calls the LLM.
|
||||
|
||||
## D&D Spell Validators
|
||||
|
||||
All four validators receive `dnd.SpellList` directly. The shape validator
|
||||
rejects missing or empty spell fields and empty reference lists. The catalog
|
||||
validator defers when shape is invalid, then checks every non-empty spell name
|
||||
against the immutable effective SRD and overlay catalog. It accepts normalized
|
||||
canonical names and aliases without rewriting the artifact; unknown names
|
||||
reject the complete result with bounded, stable index/name diagnostics. The
|
||||
source-reference validator defers malformed shapes, validates every cited
|
||||
range, and reports all range defects through a bounded aggregate while
|
||||
preserving `invalid_source_refs`. The relatedness validator resolves all cited
|
||||
ranges through the shared document-order traversal, then warns when a normalized
|
||||
consecutive spell-name token sequence is absent from the cited source text.
|
||||
Invalid shape
|
||||
or cited ranges produce no relatedness warnings; the shape and source-reference
|
||||
validators own those defects.
|
||||
|
||||
These validators are deterministic. Shape, source-reference, and relatedness
|
||||
each expose a local semantic `policy` checkpoint fingerprint. The catalog
|
||||
validator instead exposes its effective catalog digest as its semantic
|
||||
checkpoint identity and does not add a separate policy fingerprint. Their
|
||||
selectable keys and production order are defined in
|
||||
[Configuration](../config.md#implemented-production-validators); their durable
|
||||
payload rules are defined in the
|
||||
[artifact contract](../integrations/dnd-spell-artifacts.md).
|
||||
|
||||
## D&D NPC Validators
|
||||
|
||||
NPC shape validation checks required strings, arrays, and source-reference
|
||||
shape. The source-reference validator defers malformed shapes, checks
|
||||
current-document identity, unit existence, and range ordering, and reports all
|
||||
defects through bounded aggregates. Source relatedness uses the shared
|
||||
document-order traversal and normalized consecutive-token matching, emitting at
|
||||
most one bounded warning per record when neither the canonical name nor an
|
||||
alias occurs near its cited text. Invalid shape or cited ranges produce no
|
||||
relatedness warnings. Normalize identity validation checks deterministic IDs,
|
||||
canonical names, aliases, and cross-record ownership or canonical collisions.
|
||||
All are deterministic and expose the policy fingerprints used by the
|
||||
production chains.
|
||||
|
||||
## D&D Combat Validators
|
||||
|
||||
Combat shape validation owns required arrays, strings, nullable values, positive
|
||||
rounds, and supported enums. Combat source-reference validation defers invalid
|
||||
shape, checks source identity, unit existence, and range order, and reports all
|
||||
defects through bounded aggregates. Combat source-relatedness defers invalid
|
||||
shape or ranges, uses the shared traversal to combine overlapping cited units
|
||||
in document order, and emits at most one bounded advisory warning per turn for
|
||||
unrelated actors or declaration text.
|
||||
Actors use normalized consecutive-token matching; declarations retain the
|
||||
minimum four-rune token heuristic. The normalized-invariants
|
||||
validator owns display normalization, comparison-unique targets, canonical
|
||||
source-reference order, chronology, and exact duplicate identity; it defers
|
||||
shape and source-reference failures. All four validators are deterministic and
|
||||
expose local policy fingerprints. The D&D registrar orders them after generic
|
||||
JSON and response-schema validation at extraction and normalization.
|
||||
|
||||
## Production Registration
|
||||
|
||||
Production composition occurs through family registrars. The CLI allocates one
|
||||
complete framework registry set and one LLM asset registry. It invokes
|
||||
`internal/modules/generic/register`,
|
||||
`internal/modules/seriatim/register`, and `internal/modules/dnd/register` in
|
||||
that order, then exposes the matching catalog for resolution. The generic and
|
||||
Seriatim registrars own their production leaf registrations. The D&D registrar
|
||||
owns D&D leaf registrations, typed spell, NPC, and combat default-validator
|
||||
chains, typed append-order specializations, and D&D prompt/schema asset
|
||||
collection. Its registration helpers group module, validator, prompt-asset, and
|
||||
chain composition while retaining artifact-specific merge and clone behavior in
|
||||
the registrar.
|
||||
|
||||
Concrete implementation packages do not import generic implementation
|
||||
packages directly. A concrete family's `register` package is its composition
|
||||
point for specializing reusable generic implementations, while the generic
|
||||
registrar composes only generic children.
|
||||
|
||||
Core and framework production packages do not import production extensions.
|
||||
CLI production code is the sole application composition root for extensions
|
||||
and imports only exact family registrar packages. Other production packages,
|
||||
including commands and newly introduced package trees, do not import module
|
||||
packages directly. Compatibility tests in the CLI, core, and framework trees
|
||||
may import roots and implementation leaves directly. Other non-module tests do
|
||||
not receive that exemption. White-box tests within module families retain the
|
||||
production family boundaries. `internal/modules/integration` is test
|
||||
infrastructure: its black-box tests may compose multiple families, but it is
|
||||
not a production module family or production dependency target.
|
||||
|
||||
## Adding An Extension
|
||||
|
||||
When adding a production module or validator:
|
||||
|
||||
1. implement the stage or validator contract and package-local key;
|
||||
2. expose and test its spec, constructor, and registration function;
|
||||
3. keep format or domain parsing inside the concrete package;
|
||||
4. add package-owned prompt/schema assets when the extension is LLM-backed;
|
||||
new LLM-backed D&D extraction modules must follow the stable-to-variable
|
||||
prompt ordering, shared-asset ownership, and cache-boundary policy in
|
||||
[LLM Runtime](llm.md#dd-extraction-prompt-ordering-and-cache-boundaries), or
|
||||
document the implemented exception and its evidence there;
|
||||
5. register it through its package-family registrar and add a default chain
|
||||
there only when production policy requires one;
|
||||
6. add resolution and composition coverage for capabilities, options,
|
||||
references, and validation behavior;
|
||||
7. update the selectable-key catalog in [Configuration](../config.md), the
|
||||
relevant external contract, this inventory, and maintained examples when
|
||||
user-visible behavior changes.
|
||||
|
||||
Do not add the extension to `docs/development.md`; that file routes by task and
|
||||
does not inventory implementations.
|
||||
|
||||
## Tests To Inspect
|
||||
|
||||
- Package-local `*_test.go` files under the module or validator being changed.
|
||||
- `internal/framework/pipeline/typed_resolution_test.go`: typed registry, spec,
|
||||
and heterogeneous artifact composition.
|
||||
- `internal/framework/pipeline/profile_test.go`: framework binding defaults and
|
||||
profile resolution.
|
||||
- `internal/cli/production_contract_test.go`: production catalog, config
|
||||
resolution, and composition smoke coverage.
|
||||
- `internal/cli/example_contract_test.go`: maintained example ownership.
|
||||
- `internal/framework/promptfs/*_test.go` and
|
||||
`internal/modules/dnd/shared/*_test.go`: shared prompt and reference assembly.
|
||||
- `internal/modules/integration/*_test.go`: black-box composition across
|
||||
production extension domains.
|
||||
# Module Internals
|
||||
|
||||
This guide owns the mechanics for implementing and registering production
|
||||
modules. [Configuration](../config.md) owns selectable keys, binding syntax,
|
||||
reference configuration, and default validator chains. Durable input and output
|
||||
shapes belong in [integration contracts](../integrations/).
|
||||
|
||||
The D&D family has additional shared conventions and domain-specific
|
||||
exceptions. See [D&D Module Internals](dnd.md) rather than adding them here.
|
||||
|
||||
## Module Boundary
|
||||
|
||||
A module is a typed implementation registered for one pipeline stage. Its
|
||||
`ModuleSpec` is the public-to-the-framework declaration of its stable key,
|
||||
stage, execution class, required and provided capabilities, artifact kind, and
|
||||
accepted reference slots. The execution class states whether a module is
|
||||
`deterministic` or `llm_backed`; registries retain it for catalog inspection and
|
||||
resolved-pipeline debug data without constructing the module. The framework
|
||||
uses the declaration to resolve a configured binding before it builds the
|
||||
implementation. After selection, the resolver applies profile inheritance only
|
||||
to bindings whose declared execution class is `llm_backed` and rejects a
|
||||
binding-specific profile on a deterministic module. The user-facing precedence
|
||||
contract belongs in [Configuration](../config.md#pipelines).
|
||||
|
||||
Implementations that accept options must provide both an option validator and
|
||||
a builder. The validator is used while resolving configuration; the builder
|
||||
decodes the same options and constructs the implementation from the prepared
|
||||
`BuildRequest`. Reject unknown options in both paths. A builder receives only
|
||||
the dependencies and materialized references that the framework prepared for
|
||||
that operation, so it must not re-read configuration or files.
|
||||
|
||||
Registry helpers register the typed builder for a stage-specific registry.
|
||||
They are preferable to hand-written untyped registration because they retain
|
||||
the artifact type at the framework boundary. Registrars validate the registries
|
||||
they need, register each leaf implementation, and add any family-owned assets
|
||||
or default validator chains. They return contextual errors so production
|
||||
composition fails at startup rather than at the first run.
|
||||
|
||||
An artifact family can register an optional typed evidence projector alongside
|
||||
its codec. The projector returns defensive copies of the artifact's direct
|
||||
generic source references and must use the codec's exact Go type. It does not
|
||||
interpret surrounding context or publish files; the pipeline validates the
|
||||
capability during preparation and the output boundary owns publication. See
|
||||
the [Published Evidence Context contract](../integrations/evidence-context.md)
|
||||
for the durable result.
|
||||
|
||||
## Production Composition
|
||||
|
||||
Production composition is intentionally split by family:
|
||||
|
||||
- The generic registrar provides the unit chunker, generic JSON validators,
|
||||
and JSON output encoder.
|
||||
- The Seriatim registrar provides the transcript input adapter. Its external
|
||||
input behavior is defined by the [Seriatim contract](../integrations/seriatim.md).
|
||||
- The D&D registrar provides its codecs, extractors, mergers, normalizers,
|
||||
validators, prompt assets, fallback profile asset, and default chains. Its behavioral conventions
|
||||
are documented in [D&D Module Internals](dnd.md).
|
||||
|
||||
The CLI owns the composition that invokes these registrars. A module package
|
||||
may register its own family but must not assemble the CLI or make framework
|
||||
packages depend on production extensions.
|
||||
|
||||
## Adding Or Changing A Module
|
||||
|
||||
1. Choose the pipeline stage and the typed artifact boundary. Put external
|
||||
input or durable artifact formats in the relevant integration contract,
|
||||
not in this guide or in a private LLM response type.
|
||||
2. Define a stable `ModuleSpec` with an explicit execution class, the exact
|
||||
capabilities, and reference slots needed for the operation. Model a
|
||||
producer/consumer handoff as an artifact-compatible slot; configuration
|
||||
then chooses an external file or a generated binding.
|
||||
3. Implement strict option decoding, construction, and the typed stage
|
||||
interface. Preserve caller ownership: do not retain mutable request data
|
||||
and return defensive copies where an implementation exposes stored data.
|
||||
4. Register the module through its typed registry helper and add it to the
|
||||
owning family registrar. Add a default validator chain only when that
|
||||
family owns the behavior; otherwise require an explicit compatible chain.
|
||||
5. Update the selectable-key and chain reference in
|
||||
[Configuration](../config.md#production-module-keys), the applicable
|
||||
integration contract, and focused tests. Keep the configuration document
|
||||
as the sole list of production keys and validator order.
|
||||
|
||||
## Validation And References
|
||||
|
||||
Validators operate on the value produced at their configured stage. A default
|
||||
chain is ordered behavior, not a set: JSON parsing, structural checks,
|
||||
domain-specific checks, durable-schema checks, and advisory checks may have
|
||||
different responsibilities and failure handling. The active default chains and
|
||||
override rules are maintained in
|
||||
[Configuration](../config.md#production-validator-keys-and-default-chains).
|
||||
|
||||
Reference slots are part of the module specification. They describe the
|
||||
accepted artifact kind, media type, size, and whether a binding is required;
|
||||
the framework validates those constraints before construction. An external
|
||||
reference is materialized during preparation. A generated reference is a
|
||||
compatible normalized artifact handed from an earlier pipeline step at
|
||||
operation time. The configuration reference rules, including precedence and
|
||||
ordered-handoff requirements, are maintained in
|
||||
[Configuration](../config.md#references-and-ordered-handoffs).
|
||||
|
||||
## Focused Verification
|
||||
|
||||
Exercise the leaf implementation and its registration path when changing a
|
||||
module. Registry and registrar tests cover duplicate keys, required registries,
|
||||
and typed construction; pipeline resolution tests cover capabilities, options,
|
||||
and reference compatibility. Domain packages should additionally test their
|
||||
codecs, validators, normalizers, and any integration handoffs they own.
|
||||
|
||||
Run the affected package tests while iterating. The complete module suite is:
|
||||
|
||||
~~~sh
|
||||
go test ./internal/modules/...
|
||||
~~~
|
||||
|
||||
@@ -1,169 +1,58 @@
|
||||
# Internal Overview
|
||||
|
||||
This document inventories the implemented Notarius components. Normative
|
||||
This document is the implemented component map for Notarius. Normative
|
||||
boundaries and dependency direction belong in
|
||||
[Architecture](../policy/architecture.md); external behavior belongs in the
|
||||
[CLI](../cli.md), [Configuration](../config.md),
|
||||
[Architecture](../policy/architecture.md). User and operator contracts belong
|
||||
in the [CLI](../cli.md), [Configuration](../config.md),
|
||||
[Operations](../operations.md), and [integration contracts](../integrations/).
|
||||
|
||||
## Execution Path
|
||||
|
||||
`cmd/notarius` delegates to `internal/cli`, the production composition root.
|
||||
The CLI loads configuration, builds the production catalogs and runtime
|
||||
collaborators, invokes `internal/framework/pipeline`, and places the logical
|
||||
output files returned by the runner. Cache and debug collaborators are supplied
|
||||
at this boundary.
|
||||
~~~
|
||||
cmd/notarius -> internal/cli -> configuration and production composition
|
||||
-> internal/framework/pipeline -> logical output files
|
||||
-> internal/cli -> durable output and optional state/debug data
|
||||
~~~
|
||||
|
||||
Resolution produces a fixed ordered workflow of steps and globally unique,
|
||||
sorted artifact lanes. Preparation constructs the complete module and validator
|
||||
set before the runner receives source bytes. Source parsing and chunking are
|
||||
serial. Each step then uses a bounded run-wide extraction pool followed by
|
||||
serial per-lane merge and normalize continuations. A step barrier prevents
|
||||
later consumers from starting until all earlier lanes are terminal and their
|
||||
required normalized artifacts have crossed the typed handoff.
|
||||
The CLI is the application boundary: it discovers configuration, composes
|
||||
production registries and runtime collaborators, invokes the framework, and
|
||||
places returned files. The framework resolves and prepares a fixed extraction
|
||||
pipeline, then returns logical results without owning process behavior or
|
||||
physical state roots.
|
||||
|
||||
## Application Boundary
|
||||
## Components
|
||||
|
||||
| Package | Implemented responsibility |
|
||||
| --- | --- |
|
||||
| `cmd/notarius` | Executable entry point and process exit delegation. |
|
||||
| `internal/cli` | Command parsing, config discovery, package-family registrar invocation, LLM client construction, reference materialization, state collaborator setup, durable writes, and user-facing results. |
|
||||
|
||||
## Core Packages
|
||||
|
||||
| Package | Implemented responsibility |
|
||||
| --- | --- |
|
||||
| `internal/core/artifacts` | Run-manifest and provenance models. |
|
||||
| `internal/core/config` | Defaults, YAML parsing, environment overrides, validation, redaction, and effective pipeline resolution. |
|
||||
| `internal/core/debugbundle` | Explicit per-run debug-bundle allocation and redacted summary writing. |
|
||||
| `internal/core/fileio` | Generic confined atomic file and JSON writes with caller-selected permissions. |
|
||||
| `internal/core/source` | Generic source documents, units, chunks, canonical references, validation, deterministic source digests, and independent metadata materialization. |
|
||||
|
||||
## Framework Packages
|
||||
|
||||
| Package | Implemented responsibility |
|
||||
| --- | --- |
|
||||
| `internal/framework/contracts` | Source-stage contracts plus artifact identity, schema, serialized representation, codec, validator, reference, output, and structured-completion interfaces and data types. |
|
||||
| `internal/framework/pipeline` | Module and artifact-codec registries, ordered-step and generated-reference resolution, option validation, profile resolution, capability checks, external reference materialization, complete pipeline preparation, typed handoff, retries, orchestration, warnings, checkpoint decisions, and manifest population. |
|
||||
| `internal/framework/validate` | Shared validator decision and cardinality helpers. |
|
||||
| `internal/framework/llm` | Scriptorium-backed structured completions, prompt/schema registration, scheduling, profile recording, and secret redaction. |
|
||||
| `internal/framework/promptfs` | Builds module prompt filesystems from module-owned and caller-provided shared prompt assets. |
|
||||
| `internal/framework/checkpoint` | Root-based checkpoint loading, recording, identity, and payload serialization. |
|
||||
| `internal/framework/chunkplan` | Source-addressed chunk-plan filesystem storage, envelope validation, and atomic publication. |
|
||||
| `internal/framework/debug` | Root-based framework and LLM debug recording. |
|
||||
|
||||
Framework contracts provide typed artifact, provenance-wrapper, chunk-validator,
|
||||
serialized-validator, and
|
||||
typed-validator interfaces. The runner owns handoff provenance, validation
|
||||
sequencing, rejection handling, checkpoint and debug boundaries, and final
|
||||
manifest assembly.
|
||||
|
||||
Artifact registries support heterogeneous typed extraction entries and
|
||||
kind-specific merger, normalizer, and validator variants. Resolution derives a
|
||||
lane's kind from its extractor, requires the matching codec, verifies exact Go
|
||||
type equality across the lane, and records schema identity in the resolved lane
|
||||
and pipeline digest. Registry entries carry separate option-validation and
|
||||
run-local construction closures. Preparation injects shared dependencies and
|
||||
constructs input, chunk, validators, ordered lanes, and output before source
|
||||
parsing. Production modules use strict construction-time option decoding, and
|
||||
LLM-backed modules retain the injected shared client. The D&D family registers
|
||||
the canonical `dnd/spell-list`, `dnd/npc-list`, and `dnd/combat-turn-list`
|
||||
codecs, typed spell, NPC, and combat extractors and normalizers, validators,
|
||||
plus kind-specific generic merge strategies; generic JSON validators use the
|
||||
serialized-validation contract. The runner executes lanes through
|
||||
private exact-type-checked closures, coordinates extract results independently
|
||||
of completion timing, and serializes artifacts only through their codec at
|
||||
checkpoint, debug, and output boundaries.
|
||||
|
||||
## Production Extensions
|
||||
|
||||
The canonical catalogs of user-selectable
|
||||
[module](../config.md#implemented-production-modules) and
|
||||
[validator](../config.md#implemented-production-validators) keys are in
|
||||
Configuration. The implemented module packages are:
|
||||
|
||||
| Package | Implemented responsibility |
|
||||
| --- | --- |
|
||||
| `internal/modules/seriatim/input/transcript` | Parses the supported Seriatim transcript format into the generic source model. |
|
||||
| `internal/modules/generic/chunk/units` | Splits ordered source units by unit count and overlap. |
|
||||
| `internal/modules/dnd/chunk/scenes` | Produces contiguous D&D scene chunks from structured model output. |
|
||||
| `internal/modules/dnd` | Owns the canonical D&D spell-list, spell-cast, NPC-list, NPC, relationship, combat-turn-list, combat-turn, and combat-action artifact types. |
|
||||
| `internal/modules/dnd/codec/spells` | Strictly decodes and stably encodes the durable D&D spell-list representation. |
|
||||
| `internal/modules/dnd/codec/npcs` | Strictly decodes and stably encodes the durable D&D NPC-list representation. |
|
||||
| `internal/modules/dnd/codec/combatturns` | Strictly decodes and stably encodes the durable D&D combat-turn-list representation. |
|
||||
| `internal/modules/dnd/extract/spells` | Maps private structured model output to canonical source-grounded D&D spell lists. |
|
||||
| `internal/modules/dnd/extract/npcs` | Maps private structured model output to canonical source-grounded D&D NPC lists. |
|
||||
| `internal/modules/dnd/extract/combatturns` | Maps private structured model output to source-grounded D&D combat-turn candidates and preserves chronology and invalid candidate values for validators. |
|
||||
| `internal/modules/dnd/normalize/combatturns` | Canonicalizes and orders merged combat turns, applies exact NPC identity matches, and collapses only exact valid-evidence duplicates. |
|
||||
| `internal/modules/dnd/validate/combatturns` | Provides deterministic shape, source-reference, source-relatedness, and normalized-invariant validation for the production combat chains. |
|
||||
| `internal/modules/dnd/npcs/registry` | Resolves validated normalized NPC references into immutable grounding data and exact identity lookup. |
|
||||
| `internal/modules/dnd/npcs/identity` | Owns Unicode-aware NPC identity, ID derivation, and registry collision validation. |
|
||||
| `internal/modules/dnd/spells/catalog` | Embeds and validates the versioned D&D 5e 2014 SRD catalog, composes optional overlays, and provides immutable effective lookup. |
|
||||
| `internal/modules/generic/merge/appendorder` | Combines accepted extraction results in chunk order. |
|
||||
| `internal/modules/generic/normalize/noop` | Preserves accepted merged output. |
|
||||
| `internal/modules/dnd/normalize/spells` | Canonicalizes catalog-backed spell names and exact source references, conservatively collapses duplicate casts, and reports deterministic warnings and independently scoped catalog checkpoint identity. |
|
||||
| `internal/modules/dnd/normalize/npcs` | Consolidates NPC records deterministically by identity and aliases, rewrites unambiguous relationship targets, and reports bounded warnings. |
|
||||
| `internal/modules/generic/output/json` | Encodes manifests, lane payloads, warnings, and rejections as logical JSON files. |
|
||||
|
||||
`internal/modules/dnd/shared` owns reusable D&D prompt fragments,
|
||||
reference declarations, prompt input assembly, source-unit reference helpers,
|
||||
and bounded diagnostics under `internal/modules/dnd/shared/diagnostics`.
|
||||
The shared NPC grounding fragment is mounted for D&D prompts and is owned by
|
||||
this package. Domain-neutral prompt filesystem composition lives in
|
||||
`internal/framework/promptfs`.
|
||||
|
||||
The `dnd/npcs/registry` package owns the optional `npcs` registry boundary.
|
||||
External references are strictly decoded and identity-validated during
|
||||
preparation; generated references are decoded and identity-validated at the
|
||||
ordered step handoff. Both paths emit canonical registry JSON to operation-time
|
||||
spell and combat prompt or normalization requests. The framework records
|
||||
generated identity and bounded producer provenance, while the raw external
|
||||
reference remains independently tracked by pipeline provenance. An absent
|
||||
registry is represented only by the empty prompt value `{"npcs":[]}`. Spell
|
||||
and combat consumers use this shared boundary without changing their public
|
||||
module contracts.
|
||||
|
||||
Generic validators under `internal/modules/generic/validate` provide
|
||||
unconditional test decisions, JSON syntax validation, and JSON Schema
|
||||
validation. D&D spell validators under `internal/modules/dnd/validate/spells`
|
||||
consume the canonical spell-list type directly to provide shape,
|
||||
effective-catalog, source-reference, and source-relatedness decisions.
|
||||
|
||||
Production composition is grouped behind package-family registrars, and every
|
||||
implemented production extension uses its domain-first tree:
|
||||
|
||||
| Package | Implemented responsibility |
|
||||
| --- | --- |
|
||||
| `internal/modules/generic/register` | Registers domain-neutral chunk, merge, normalize, output, and validator implementations. |
|
||||
| `internal/modules/seriatim/register` | Registers the Seriatim input adapter. |
|
||||
| `internal/modules/dnd/register` | Registers D&D modules, validators, default validator policy, and prompt/schema assets. |
|
||||
|
||||
The CLI allocates the framework registries and asset registry, then invokes
|
||||
these registrars in generic, Seriatim, and D&D order.
|
||||
|
||||
Implementation details for all production extensions are in
|
||||
[Module Internals](modules.md).
|
||||
|
||||
## Run-State Components
|
||||
|
||||
| Surface | Implemented owners | Internal purpose |
|
||||
| Area | Implemented owners | Responsibility |
|
||||
| --- | --- | --- |
|
||||
| Durable output | Output module, pipeline runner, and CLI writer | Return logical consumer files and place them for a run. |
|
||||
| Cache checkpoints | `internal/framework/checkpoint` and `internal/cli` | Validate and serialize reusable extract, merge, and normalize outcomes, including ordered-step scope and generated-artifact dependency decisions. |
|
||||
| Chunk-plan cache | `internal/framework/chunkplan` and `internal/cli` | Persist and select source-addressed plans before framework materialization. |
|
||||
| Debug bundles | `internal/core/debugbundle`, `internal/framework/debug`, and pipeline instrumentation | Persist redacted summaries and application-owned traces. |
|
||||
| Executable and command boundary | **cmd/notarius**, **internal/cli** | Process entry, command dispatch, configuration discovery, production composition, runtime collaborator setup, durable file placement, and user-facing reporting. |
|
||||
| Configuration | **internal/core/config** | Defaults, strict YAML parsing, environment overrides, structural validation, effective resolution, redaction, and resolved-composition summaries. |
|
||||
| Generic models | **internal/core/source**, **internal/core/artifacts**, **internal/framework/contracts** | Source documents and chunks, manifests and provenance, plus typed artifact, reference, validation, output, and structured-completion contracts. |
|
||||
| Pipeline framework | **internal/framework/pipeline** | Registries, profile and reference resolution, typed preparation, validation, retry coordination, ordered execution, handoff, and result assembly. |
|
||||
| LLM and prompt runtime | **internal/framework/llm**, **internal/framework/promptfs** | Provider-neutral structured completions, scheduling, profile recording, prompt assets, schema registration, and credential-shaped-value redaction. |
|
||||
| Runtime state | **internal/core/fileio**, **internal/core/debugbundle**, **internal/framework/checkpoint**, **internal/framework/chunkplan**, **internal/framework/chunkmap**, **internal/framework/debug** | Confined atomic files, debug bundles, checkpoint and chunk-plan state, accepted chunk maps, and pipeline-facing debug recording. |
|
||||
| Production extensions | **internal/modules/generic**, **internal/modules/seriatim**, **internal/modules/dnd** | Domain-neutral extensions, Seriatim input support, and D&D extraction families registered into the production catalog. |
|
||||
|
||||
Physical layout, cleanup, recovery, and sensitive-data handling are defined
|
||||
in [Operations](../operations.md). Concrete modules receive recorder
|
||||
interfaces and request data, not physical state roots.
|
||||
Generic core and framework packages do not depend on production extensions.
|
||||
Concrete extensions depend inward on their contracts and are registered only at
|
||||
the CLI composition boundary.
|
||||
|
||||
## Focused Documentation
|
||||
|
||||
- [Pipeline Internals](pipeline.md): resolution, execution, validation, retries,
|
||||
checkpoint/debug hooks, and result assembly.
|
||||
- [Module Internals](modules.md): production modules, validators, assets,
|
||||
registration, and the contributor recipe for adding an extension.
|
||||
- [LLM Runtime](llm.md): structured completion contracts, Scriptorium adapter,
|
||||
assets, scheduling, profile recording, and redaction.
|
||||
- [Configuration Internals](configuration.md): loading, validation, effective
|
||||
resolution, redaction, and resolved-composition identity.
|
||||
- [CLI Internals](cli.md): command dispatch, production composition, run
|
||||
orchestration, and terminal reporting.
|
||||
- [Pipeline Internals](pipeline.md): resolution, preparation, execution,
|
||||
validation, typed handoff, and framework state hooks.
|
||||
- [Run State Internals](state.md): output, cache, debug collaborator
|
||||
composition, and path safety.
|
||||
- [LLM Runtime](llm.md): structured completion, scheduling, prompt assets,
|
||||
profiles, and secret handling.
|
||||
- [Module Internals](modules.md): generic extension registration, module
|
||||
construction, validation, and reference mechanics.
|
||||
- [D&D Module Internals](dnd.md): shared D&D extractor conventions, generated
|
||||
reference projections, and lane-specific exceptions. Durable D&D and
|
||||
Seriatim data shapes remain in the [integration contracts](../integrations/).
|
||||
|
||||
Use this map to find an owner, then read the focused document and its tests
|
||||
before changing behavior.
|
||||
|
||||
@@ -1,440 +1,185 @@
|
||||
# Pipeline Internals
|
||||
|
||||
The implemented resolver and runner live in `internal/framework/pipeline`.
|
||||
Their fixed workflow and ownership boundaries are defined by
|
||||
[Architecture](../policy/architecture.md#system-shape). Configuration fields,
|
||||
defaults, and selectable keys are defined in
|
||||
[Configuration](../config.md#pipelines).
|
||||
This document describes the framework-owned pipeline mechanics in
|
||||
**internal/framework/pipeline**. [Configuration](../config.md) owns selectable
|
||||
profiles, bindings, and retry settings; [Operations](../operations.md) owns
|
||||
state lifecycle and recovery; and the [integration contracts](../integrations/)
|
||||
own durable output shapes. Concrete production extensions are covered by
|
||||
[Module Internals](modules.md).
|
||||
|
||||
Resolution fixes the ordered steps, selected lanes, and all stage bindings;
|
||||
preparation constructs every selected implementation before the runner begins
|
||||
source work. After serial input parsing and plan selection or generation, the
|
||||
runner materializes chunks and executes one step at a time. Within a step,
|
||||
extract work uses one bounded run-wide worker pool in chunk-first, lane-second
|
||||
order. Each lane's merge and normalize operations remain serial, and lanes in
|
||||
the same step may overlap once their extracts are terminal. A later step cannot
|
||||
start across its barrier until every earlier lane is terminal and each required
|
||||
generated artifact has been accepted and handed off.
|
||||
## Boundary
|
||||
|
||||
## Resolution
|
||||
The pipeline framework accepts a resolved composition, registries, shared
|
||||
dependencies, input bytes, and state/debug collaborators. It returns logical
|
||||
output files, normalized artifacts, recorded rejections and warnings, manifest
|
||||
provenance, and checkpoint decisions. The CLI owns process arguments,
|
||||
configuration discovery, physical roots, and placement of returned output
|
||||
files.
|
||||
|
||||
`internal/core/config.Config.Resolve` validates the loaded configuration,
|
||||
selects the named profile, applies the runtime inputs supplied by the CLI, and
|
||||
calls `pipeline.ResolvePipeline`.
|
||||
The framework has one fixed shape:
|
||||
|
||||
`ResolvePipeline`:
|
||||
~~~
|
||||
input -> chunk -> extract -> merge -> normalize -> output
|
||||
~~~
|
||||
|
||||
1. selects the explicit ordered steps, or creates the implicit `default` step
|
||||
from the legacy top-level `artifacts` map;
|
||||
2. selects and sorts artifact lanes within each step while enforcing global lane
|
||||
identity;
|
||||
3. completes omitted bindings using the documented configuration defaults;
|
||||
4. looks up each module and validator spec without constructing it;
|
||||
5. for a typed extractor, derives its artifact kind, requires the codec, and
|
||||
selects exact-type merger, normalizer, and validator variants;
|
||||
6. checks required and provided capabilities in workflow order;
|
||||
7. resolves external and generated target-aware reference bindings and
|
||||
validates producer order, consumer slot declarations, and artifact-kind
|
||||
compatibility;
|
||||
8. validates each selected module and validator option set through its registry
|
||||
entry; and
|
||||
9. calculates a digest over the resolved structure, including step order, step
|
||||
IDs, lane membership, generated topology, producer and consumer identities,
|
||||
typed artifact kind and schema identity, and the effective validator policy
|
||||
in its resolved execution order.
|
||||
Input and chunking are pipeline-wide. A selected artifact lane owns extract,
|
||||
merge, and normalize; output aggregates the terminal lane outcomes. A pipeline
|
||||
is an ordered list of steps, not an arbitrary workflow graph.
|
||||
|
||||
Resolution returns a `ResolvedPipeline` containing ordered steps, lanes,
|
||||
concrete bindings, validator chains, reference targets, and the digest. It does
|
||||
not read external reference bytes or construct runtime modules. CLI lane and
|
||||
reference selector syntax is defined in the [CLI reference](../cli.md#run).
|
||||
## Resolve, Materialize, Prepare
|
||||
|
||||
The digest includes each resolved step's ID and lane membership, generated
|
||||
producer/consumer topology, and each validator chain's stage, lane, owning
|
||||
module, ordered validator bindings, execution classes, targets, and artifact
|
||||
kinds. Changing step order, a dependency, a default chain, or an explicit
|
||||
override therefore changes pipeline identity whenever it changes effective
|
||||
execution policy.
|
||||
Resolution turns a configured pipeline profile into a **ResolvedPipeline**.
|
||||
It normalizes the pipeline and lane identities, applies stage defaults, selects
|
||||
requested lanes where that is supported, resolves validator chains, checks
|
||||
module capabilities and typed artifact compatibility, validates options, and
|
||||
assigns a deterministic resolved-composition digest. The resolved pipeline
|
||||
contains bindings and declared reference targets, not external reference bytes.
|
||||
After selection, the resolver applies command, binding, and pipeline profile
|
||||
precedence to LLM-backed bindings and validators only; prompt defaults remain
|
||||
an empty resolved binding profile. Deterministic bindings remain profile-free.
|
||||
These effective values are part of the digest, so execution and checkpoint
|
||||
consumers do not repeat profile inheritance.
|
||||
Configuration resolution supplies the selected profile and catalog; see
|
||||
[Configuration Internals](configuration.md).
|
||||
|
||||
## Reference Materialization
|
||||
External reference materialization happens before preparation. The materializer
|
||||
checks that each slot is declared by the selected module, resolves a file path
|
||||
relative to the correct configuration or working-directory origin, reads
|
||||
UTF-8 text, verifies media type and size limits, and retains bounded
|
||||
provenance. A generated-artifact selector remains declared but has no bytes
|
||||
until its producing step completes.
|
||||
|
||||
The CLI calls `MaterializeReferences` after resolution and before constructing
|
||||
the LLM client or running the pipeline. For external bindings, the materializer
|
||||
checks each binding against its resolved target declaration, reads and validates
|
||||
the file, and builds both a `contracts.ReferenceSet` and provenance-only
|
||||
metadata on the corresponding `ResolvedReferenceTarget`. A structured
|
||||
generated binding is declaration-only at this point: its producer bytes do not
|
||||
exist until the producer lane reaches an accepted normalized result.
|
||||
Preparation is the construction boundary. It validates the resolved shape and
|
||||
registry set, clones the resolved data, then constructs the input adapter,
|
||||
chunker, stage-local validators, every typed lane, and output encoder with
|
||||
cloned options, references, and shared dependencies. It also collects stable
|
||||
checkpoint fingerprints. Missing registrations, incompatible typed entries,
|
||||
nil implementations, and constructor failures are reported before source
|
||||
parsing or any stage operation begins.
|
||||
|
||||
Preparation delivers the materialized external set for each target through
|
||||
`pipeline.BuildRequest`: chunkers and chunk validators receive the chunk target;
|
||||
extractors and extract validators receive the lane extract target; mergers and
|
||||
merge validators receive the lane merge target; and normalizers and normalize
|
||||
validators receive the lane normalize target. Input and output builders receive
|
||||
an empty set because those stages cannot declare references. Every builder gets
|
||||
an isolated deep clone of its target set, so construction-time mutation cannot
|
||||
change another builder, the resolved pipeline, or later runtime requests.
|
||||
An output encoder can opt into source-evidence publication through its output
|
||||
policy. Preparation keeps the configured lane allowlist and active lanes
|
||||
separate, then verifies an exact typed evidence projector and registered codec
|
||||
for each active lane. The resulting private plan is immutable; lanes excluded
|
||||
by invocation filtering remain configured but do not acquire a projector for
|
||||
that run.
|
||||
|
||||
Prepared consumers do not need to be reconstructed when generated content is
|
||||
available. At the step boundary, the runner encodes the accepted producer value
|
||||
through its registered canonical codec, validates the generated bytes against
|
||||
each target slot's kind, schema, media type, and size, and clones one immutable
|
||||
reference item into the operation request. The item includes canonical digest,
|
||||
size, and bounded producer provenance but no filesystem URI. A handoff failure
|
||||
is a framework dependency error and prevents every consumer in that step from
|
||||
starting.
|
||||
## Typed Lanes And References
|
||||
|
||||
The runner continues to clone the resulting set into the chunk, extract, merge,
|
||||
or normalize request that owns the target. LLM-backed extensions may convert
|
||||
those items into named prompt inputs. Reference content remains separate from
|
||||
source evidence and source digests, whether the item came from a file or a
|
||||
generated handoff.
|
||||
Each resolved lane has one artifact kind, codec, and exact Go type. The
|
||||
framework uses private type erasure only around those typed operations; every
|
||||
handoff checks exact type and codec identity and reports incompatibility as an
|
||||
error rather than panicking. Encoding through the registered codec is the
|
||||
boundary for output, checkpoints, debug records, and generated references.
|
||||
|
||||
Binding precedence, path resolution, accepted content, and media-type behavior
|
||||
are configuration contracts; see [Configuration](../config.md#pipelines).
|
||||
Durable provenance is defined in the
|
||||
[JSON output contract](../integrations/json-output.md#manifestjson), while
|
||||
runtime sensitive-data handling belongs in [Operations](../operations.md).
|
||||
Reference targets are stage- and lane-specific. External reference bytes are
|
||||
cloned into the operation request. Generated references are built at the next
|
||||
step boundary from exactly one accepted normalized producer output. The
|
||||
framework decodes and re-encodes that output with the registered producer
|
||||
codec, checks its complete schema and media identity, and records a content
|
||||
digest plus bounded producer provenance. A missing, ambiguous, invalid, or
|
||||
incompatible producer prevents the consumer step from starting.
|
||||
|
||||
## Registries And Specs
|
||||
## Execution And Ordering
|
||||
|
||||
`pipeline.Registries` holds option validators and run-local builders used during
|
||||
resolution and preparation.
|
||||
`pipeline.ModuleCatalog` exposes their specs during configuration validation and
|
||||
resolution. Separate registries exist for every stage and for validators;
|
||||
`ValidatorChainRegistry` stores production default-chain mappings. Both
|
||||
containers also carry an `ArtifactCodecRegistry`. Generic registration records
|
||||
one codec per stable artifact kind, validates its schema metadata and JSON
|
||||
Schema, retains the exact schema digest and Go type, and safely encodes or
|
||||
decodes framework-erased values with typed errors on incompatibility.
|
||||
The runner validates its input, installs no-op state collaborators when none
|
||||
were supplied, and serially performs source parsing and chunk-plan selection.
|
||||
An accepted plan is materialized into source-addressed chunks and passes the
|
||||
configured chunk validators before any lane runs. A chunk rejection is a
|
||||
recorded pipeline outcome: lanes do not start, but the output stage can encode
|
||||
the terminal result.
|
||||
|
||||
Typed extractor entries are keyed by module key and declare one artifact kind.
|
||||
Merger, normalizer, and typed-validator variants are keyed by module or
|
||||
validator key plus artifact kind. Chunk and serialized validators occupy
|
||||
separate target namespaces; serialized registrations declare whether they
|
||||
support chunks, artifacts, or both. Duplicate variants and exact Go-type
|
||||
mismatches are rejected deterministically.
|
||||
For each ordered step, the runner first builds generated reference sets from
|
||||
the accepted normalized outputs of earlier steps. It then executes the step's
|
||||
lanes. Later steps do not begin until the current step is terminal and its
|
||||
generated handoffs have succeeded.
|
||||
|
||||
Lane-sensitive merger and normalizer spec discovery always supplies the
|
||||
extractor's artifact kind, so variants under one reusable key may declare
|
||||
different capabilities and reference slots. Kind-neutral registry inspection
|
||||
selects the first registered artifact kind in sorted order.
|
||||
Within a step, the lane engine dispatches extraction jobs in deterministic
|
||||
chunk-first, lane-second order to a bounded worker group. When all extraction
|
||||
jobs for one lane are terminal, a bounded continuation group can run that
|
||||
lane's merge and normalize work while extraction for other lanes continues.
|
||||
The framework does not create an unbounded goroutine per chunk or lane.
|
||||
|
||||
Production composition registers the D&D spell-list codec and typed extractor,
|
||||
matching typed merge, normalize, and semantic-validator variants, and
|
||||
serialized JSON validators. Every artifact lane resolves through the typed
|
||||
registries and a matching codec.
|
||||
Completion timing does not determine public results. The coordinator restores
|
||||
lane and chunk order before merging results, and selects a framework error by
|
||||
stable stage, lane, and chunk position. A validator rejection records a lane
|
||||
outcome without cancelling unrelated work. A framework error or parent
|
||||
cancellation cancels derived work, prevents queued work from starting, waits
|
||||
for started workers, and prevents output encoding.
|
||||
|
||||
A `ModuleSpec` declares its stage plus required and provided capabilities.
|
||||
Chunk, extract, merge, and normalize specs may also declare reference slots.
|
||||
Registry implementations defensively copy spec metadata, reject duplicate keys,
|
||||
and verify that a constructed implementation reports the registered key.
|
||||
Builder registrations accept `ModuleDependencies` and cloned configuration
|
||||
options through one `BuildRequest`. Builders decode those options and retain
|
||||
typed values or injected dependencies in the constructed implementation.
|
||||
Extractors declare their artifact kind, and merger, normalizer, and validator
|
||||
resolution selects the matching typed variant.
|
||||
## Validation, Retries, And Output
|
||||
|
||||
A `ValidatorSpec` declares a validator key and execution class. Resolution uses
|
||||
the execution class to reject incompatible profile bindings before execution.
|
||||
The current production catalog and default chain are listed only in
|
||||
[Configuration](../config.md#implemented-production-validators).
|
||||
Every chunk, extract, merge, and normalize candidate passes its resolved
|
||||
validator chain. Validators receive immutable canonical input appropriate to
|
||||
their target: chunks, typed values, or serialized codec bytes. They may
|
||||
approve, approve with warnings, reject, or fail. A rejection is an ordinary
|
||||
pipeline result; a validator error is a framework error.
|
||||
|
||||
## Preparation And Runner Boundary
|
||||
The runner applies the binding's retry policy around a stage operation and its
|
||||
complete validation chain. It preserves warnings only from the final accepted
|
||||
or rejected attempt. Cancellation stops retries. Normalizer-specific retry
|
||||
directives consume this same budget and validate any final safe fallback through
|
||||
the normalizer chain.
|
||||
|
||||
`pipeline.Prepare` receives a resolved pipeline, the registries, and shared
|
||||
module dependencies. It constructs input; chunk and its validators; every
|
||||
step's lane extract, merge, and normalize modules and validator chains in
|
||||
resolved order; then output. It stops at the first error with pipeline, step,
|
||||
stage, lane, module, and validator context as applicable. It never invokes an
|
||||
operation method. Generated references are not available during preparation;
|
||||
the operation request is the handoff boundary.
|
||||
|
||||
`PreparedPipeline` keeps private constructed executors and exposes cloned
|
||||
resolved input, chunk, lane, and output identities. Prepared components may
|
||||
implement `pipeline.CheckpointFingerprintProvider` to contribute explicit
|
||||
semantic identities to checkpoint reuse. Preparation trims and validates each
|
||||
non-secret name and value, prefixes it with the component's stage, lane,
|
||||
module, and validator scope, rejects duplicates, and retains the resulting
|
||||
sorted collection behind a defensive-copy accessor. Fingerprints must be
|
||||
stable and must not contain source content, credentials, local paths,
|
||||
timestamps, or other invocation-specific values.
|
||||
|
||||
`pipeline.RunInput` carries that prepared pipeline, raw source input, run identity and timing, optional
|
||||
session and profile metadata, a chunk-plan store and mode, a checkpoint
|
||||
execution policy, and checkpoint/debug collaborators. The runner
|
||||
parses source bytes through the already constructed input adapter. Later stage
|
||||
requests receive the generic source model; extract requests receive
|
||||
chunk-scoped input material, while chunk, merge, and normalize requests retain
|
||||
access to the original source material. Input, chunk, and output operation
|
||||
requests do not carry raw module options. The chunk request also does not carry
|
||||
an LLM client; an LLM-backed chunker receives the shared client during
|
||||
preparation. Their operation requests retain run-specific source, reference,
|
||||
profile, session, metadata, and step-handoff context as applicable. A generated
|
||||
reference is cloned into each compatible consumer request and is never exposed
|
||||
as a path.
|
||||
|
||||
Prepared lanes retain exact-type-checked erased operation closures. The runner
|
||||
uses those closures to keep each value typed through extraction, validation,
|
||||
merge, and normalization.
|
||||
|
||||
Source validation requires every unit to carry a canonical self-reference to
|
||||
its containing document and its own unit ID. Explicit clone, checkpoint, and
|
||||
debug boundaries retain that reference, and the canonical source digest covers
|
||||
it deterministically. Chunks use the same source model and carry one canonical
|
||||
reference spanning the first selected unit through the last.
|
||||
|
||||
`pipeline.RunOutput` carries the run manifest, accepted normalized serialized
|
||||
artifacts with lane and normalizer provenance,
|
||||
rejected results, warnings, checkpoint events, and logical files returned by the
|
||||
output encoder. The CLI owns debug-summary and durable filesystem writes after
|
||||
the runner returns.
|
||||
|
||||
## Execution Flow
|
||||
|
||||
The pipeline-wide coordinator owns the ordered step loop, generated-reference
|
||||
sets at each barrier, and deterministic merging of step outcomes. For one step,
|
||||
the lane engine initializes checkpoint state in lane order, dispatches bounded
|
||||
extract work, advances terminal lanes through serial merge and normalize work,
|
||||
selects failures by stable pipeline scope, and merges lane-local outcomes back
|
||||
in resolved order. Completion timing never becomes public ordering.
|
||||
|
||||
The runner:
|
||||
|
||||
1. validates its prepared input;
|
||||
2. parses the raw input with the prepared adapter and validates the generic
|
||||
source document;
|
||||
3. selects a stored plan or executes the configured chunker's `Plan` operation;
|
||||
4. canonicalizes and materializes the plan, then validates the resulting
|
||||
chunks;
|
||||
5. executes each resolved step in configuration order. For one step, it
|
||||
dispatches extract jobs in source-chunk then resolved-lane order, starts a
|
||||
bounded lane continuation when all extracts for that lane are terminal, and
|
||||
waits for every lane to become terminal;
|
||||
6. encodes and validates each accepted normalized producer artifact, then
|
||||
builds the immutable generated reference sets for the next step;
|
||||
7. invokes the prepared output encoder only after every step succeeds and
|
||||
validates its logical file results;
|
||||
8. returns the assembled manifest, outcomes, warnings, and files.
|
||||
|
||||
Within each artifact lane, it reuses the prepared extractor, merger, normalizer,
|
||||
and validators while performing these transitions:
|
||||
|
||||
1. extract once per accepted chunk and add runner-owned lane, source, and chunk
|
||||
provenance;
|
||||
2. validate each extract result and omit rejected results from merge input;
|
||||
3. skip the rest of the lane when no extract result is accepted;
|
||||
4. merge accepted extract results in their existing order;
|
||||
5. validate the merge result and skip normalization on rejection;
|
||||
6. normalize the accepted merge result;
|
||||
7. validate and append the accepted normalized result.
|
||||
|
||||
At a step barrier, a lane with no accepted normalized output is still a regular
|
||||
rejection unless a later generated binding names that lane as a required
|
||||
producer. In that case the runner raises a deterministic dependency error and
|
||||
does not start the consumer step. One accepted typed artifact may fan out to
|
||||
multiple compatible target slots. Consumers in the same step may run
|
||||
concurrently after the handoff; no work crosses the barrier early.
|
||||
|
||||
Module-provided warnings and payload warnings are promoted only from attempts
|
||||
whose results are accepted and used.
|
||||
|
||||
## Chunk Plans And Reuse
|
||||
|
||||
`Chunker.Plan` returns a `source.ChunkPlan`: the canonical source digest,
|
||||
ordered unit-ID ranges, and optional plan or range annotations. The framework
|
||||
owns plan canonicalization and materialization. It creates the generic chunks
|
||||
and therefore owns their IDs, indexes, source references, JSON content, units,
|
||||
media type, and generic metadata. Plan and range annotations are independently
|
||||
owned raw JSON and become `Chunk.PlanAnnotations` and `Chunk.Annotations`.
|
||||
|
||||
In `auto`, the runner looks up the source digest before invoking the chunker. A
|
||||
valid hit is materialized and sent through the current run's configured chunk
|
||||
validators; it does not invoke the chunk module, consume its retry budget, or
|
||||
make a chunk-stage LLM call. A missing, invalid, or unmaterializable record
|
||||
generates a candidate. `refresh` generates without lookup; `bypass` generates
|
||||
without cache access. Generated plans are published only after the full chunk
|
||||
validator chain approves them. A validator rejection is a regular rejected
|
||||
pipeline outcome and never replaces a cached plan.
|
||||
|
||||
The store is source-addressed, not pipeline-addressed. Changes to pipeline
|
||||
configuration, requested chunker, options, references, lanes, validators, or
|
||||
LLM profile do not prevent a source-digest hit. The manifest records both the
|
||||
currently requested chunker and the effective plan producer. Cache state and
|
||||
paths are configured and operated outside the runner; see
|
||||
[Configuration](../config.md#state-surfaces) and [Operations](../operations.md).
|
||||
|
||||
The extract job channel has the same capacity as the effective extract worker
|
||||
count, so dispatch applies backpressure. A fixed continuation executor prevents
|
||||
ready or checkpoint-reused lanes from creating one goroutine each. Workers and
|
||||
continuations publish lane-local results; the coordinator is the only writer of
|
||||
aggregate output and merges those results in resolved lane and source-chunk
|
||||
order.
|
||||
|
||||
## Plan Canonicalization And Chunk Materialization
|
||||
|
||||
Plan canonicalization requires canonical JSON annotations, a matching source
|
||||
digest, at least one range, existing ordered boundaries, and increasing range
|
||||
starts. Ranges may overlap or leave gaps; a chunker may impose stricter policy.
|
||||
Materialization deterministically reconstructs each range from the current
|
||||
source document, deep-clones JSON-shaped source-unit metadata, and copies
|
||||
annotations without interpreting their namespaces. Materialized chunks and
|
||||
separate materializations do not share mutable unit metadata; unsupported or
|
||||
cyclic metadata fails materialization with context.
|
||||
|
||||
Before lane execution, generic chunk validation checks the materialized chunks'
|
||||
identities, order, source references, content, media type, units, and metadata.
|
||||
No chunk checkpoint participates in plan selection: plan storage is the only
|
||||
chunk-reuse mechanism. Extract, merge, and normalize checkpoints continue to
|
||||
use materialized chunk digests as their dependencies.
|
||||
|
||||
## Validation And Retries
|
||||
|
||||
Chunk, extract, merge, and normalize results pass through the resolved validator
|
||||
chain for their stage and module. Chunk validators receive canonical chunks;
|
||||
typed validators receive the domain value; and serialized validators receive
|
||||
canonical chunk JSON or artifact codec bytes. Validators execute in resolved
|
||||
order and stop at the first error or rejection. An empty chain approves the
|
||||
result.
|
||||
|
||||
`runWithRetry` applies the effective retry policy around module execution and
|
||||
its complete validation chain. A module or validator error becomes a framework
|
||||
error when attempts are exhausted. A rejection becomes a recorded
|
||||
`RejectedOutput` when attempts are exhausted. Cancellation stops retry
|
||||
processing immediately.
|
||||
|
||||
Rejected output is a non-fatal pipeline outcome and does not advance. Warnings
|
||||
from discarded attempts are not promoted. Configuration owns retry counts and
|
||||
validator overrides; see [Module Bindings](../config.md#module-bindings).
|
||||
After terminal lane work, the runner assembles manifest provenance, normalized
|
||||
artifacts, rejections, warnings, and an optional accepted chunk map. When an
|
||||
output policy selected evidence lanes, it decodes accepted serialized normalize
|
||||
outputs through their registered codecs and invokes the prepared typed
|
||||
projectors. Rejected or absent lanes contribute nothing. This reconstruction is
|
||||
also used after normalized-checkpoint reuse, so no second typed output channel
|
||||
is retained. The runner passes the resulting owned artifact to the output
|
||||
encoder, which returns logical files and does not choose a physical directory.
|
||||
The CLI publishes those files only after the runner returns without a framework
|
||||
error. Logical file names and schemas are defined by the [output integration
|
||||
contracts](../integrations/).
|
||||
|
||||
## Checkpoint And Debug Hooks
|
||||
|
||||
The runner depends on recorder and loader interfaces, using no-op
|
||||
implementations when collaborators are absent. Each checkpointed workflow
|
||||
boundary records a running, succeeded, or failed transition. Reuse decisions
|
||||
are consulted in workflow order and accepted payloads are cloned before
|
||||
entering the normal handoff path. Typed extract, merge, and normalize
|
||||
checkpoints store codec bytes with artifact kind, schema ID, name, version and
|
||||
exact digest, and media type. Reuse compares that identity with the prepared
|
||||
codec and decodes through the codec; missing identity, mismatches, corrupt
|
||||
bytes, and decode failures become explicit reuse misses and execute the lane
|
||||
normally. Dependency fingerprints and debug content digests use the same stable
|
||||
codec bytes that cross those boundaries.
|
||||
The runner receives checkpoint and debug interfaces rather than roots. It
|
||||
records workflow transitions and reuse decisions through the supplied
|
||||
collaborators, and clones reusable artifacts before they re-enter normal typed
|
||||
handoff. Generated-reference dependencies participate in checkpoint decisions.
|
||||
Selective recomputation can require a canonical accepted normalized predecessor
|
||||
before a dependent lane starts.
|
||||
|
||||
That progressive extract, merge, and normalize reuse is the ordinary resume
|
||||
path. A lane marked as a required predecessor for selective recomputation takes
|
||||
a separate accepted-output path before extract scheduling. The loader reads the
|
||||
existing successful normalize manifest and payload by step, lane, and
|
||||
normalizer, without consulting extract or merge dependencies. It requires the
|
||||
current non-empty checkpoint identity to match, so the invocation identity
|
||||
still binds the input, resolved topology and configuration, references, runtime
|
||||
overrides, profiles, and component fingerprints.
|
||||
Debug recording is attempt-scoped and application-owned. A failure to persist
|
||||
required debug data is a framework error. State roots, persistence, reason-code
|
||||
meanings, resume, and cleanup are intentionally owned by
|
||||
[Run State Internals](state.md) and [Operations](../operations.md).
|
||||
|
||||
The runner decodes that accepted normalized artifact with the prepared codec,
|
||||
re-encodes it, and requires exact kind, schema identity and digest, media type,
|
||||
canonical bytes, content digest, and producer provenance. A valid result becomes
|
||||
a runner-owned cloned normalized output, restores only normalize-checkpoint
|
||||
warnings, and records one `accepted_artifact_reused` normalize decision. It does
|
||||
not invoke or record extract, merge, normalize, or their validators. Invalid or
|
||||
unavailable accepted state records its decision and fails the producer step;
|
||||
the dependent step never starts and the producer is not implicitly rerun.
|
||||
## Invariants To Preserve
|
||||
|
||||
Generated references add downstream dependencies containing the producer's
|
||||
artifact kind, complete schema identity, media type, canonical content digest,
|
||||
and size. Compatible accepted producer outputs may therefore feed a later step
|
||||
without re-executing the producer. Forced lanes bypass accepted-output
|
||||
hydration and execute normally. A missing, rejected, corrupt, incompatible, or
|
||||
changed producer blocks its dependent while leaving independent work eligible
|
||||
for reuse. The runner records bounded decision
|
||||
categories: `reused`, `executed`, `forced_recompute`, and
|
||||
`dependency_invalidated`. Operator meanings for the stable reason codes belong
|
||||
to [Operations](../operations.md#resume-and-selective-recompute).
|
||||
- The six fixed stages remain explicit; a pipeline is not a general DAG.
|
||||
- Resolution and preparation reject statically discoverable incompatibility
|
||||
before parsing or execution.
|
||||
- Every typed lane uses one compatible artifact kind, codec, and exact Go type.
|
||||
- Generated references come only from one earlier accepted normalized producer
|
||||
and carry canonical identity rather than an unverified value.
|
||||
- Rejections are recorded outcomes; framework errors cancel derived work and
|
||||
prevent output encoding.
|
||||
- Public ordering and selected errors are independent of goroutine completion
|
||||
order.
|
||||
- Pipeline modules receive collaborators and data, never CLI streams or
|
||||
physical output, cache, or debug roots.
|
||||
|
||||
The CLI includes prepared-component fingerprints in the run-wide checkpoint
|
||||
identity alongside resolved configuration, raw input, reference provenance,
|
||||
runtime overrides, and LLM-profile fingerprints. Module metadata is not used
|
||||
implicitly for cache identity: components opt in only with stable semantic
|
||||
values that can change accepted output. Adding or changing a component
|
||||
fingerprint intentionally produces a cold cache miss. Existing checkpoint
|
||||
schemas and paths remain unchanged.
|
||||
## Focused Tests
|
||||
|
||||
The CLI's `--recompute-step` policy forces the selected step and all transitive
|
||||
dependents, but requires accepted normalized artifacts for every unselected
|
||||
producer on which that closure depends. It changes execution policy only; it
|
||||
does not alter persistent checkpoint identity.
|
||||
- **internal/framework/pipeline/profile_test.go** and
|
||||
**typed_resolution_test.go** cover resolution, defaults, ordered steps,
|
||||
compatibility, validators, references, and resolved identity.
|
||||
- **internal/framework/pipeline/preparation_test.go** covers complete
|
||||
construction before execution and contextual construction failures.
|
||||
- **internal/framework/pipeline/references_test.go** and **handoff_test.go**
|
||||
cover external materialization, generated references, provenance, and typed
|
||||
producer checks.
|
||||
- **internal/framework/pipeline/runner_concurrency_test.go** covers bounded
|
||||
execution, ordered steps, stable error selection, rejections, and
|
||||
cancellation.
|
||||
- **internal/framework/pipeline/runner_chunk_plan_test.go**,
|
||||
**runner_typed_checkpoint_test.go**, and
|
||||
**runner_accepted_checkpoint_test.go** cover state hooks and reuse behavior.
|
||||
- **internal/framework/pipeline/runner_attempt_debug_test.go** and
|
||||
**runner_terminal_debug_test.go** cover attempt and terminal debug behavior.
|
||||
|
||||
Debug instrumentation wraps run, stage, attempt, validator, and structured LLM
|
||||
boundaries. Every executed chunk, extract, merge, and normalize attempt writes
|
||||
one terminal envelope for acceptance, validator rejection, module or validator
|
||||
error, or applicable candidate or final serialization error. The envelope
|
||||
contains its attempt-local warnings, any available candidate and rejection,
|
||||
and terminal error text; failures before a candidate exists omit that payload.
|
||||
Only LLM calls made by the module operation belong to the module attempt.
|
||||
Validator calls retain independent scopes under `validate/` and are not
|
||||
duplicated into the module envelope. A failed terminal-envelope write is a
|
||||
non-retryable framework error and is joined with any primary attempt error.
|
||||
Debug data is never used as a checkpoint source. Typed artifact debug envelopes
|
||||
are domain-neutral, redact sensitive metadata and bytes through the common
|
||||
debug policy, and record codec identity plus schema and content digests.
|
||||
|
||||
Merge and normalize attempts serialize their in-memory candidate with the
|
||||
codec's required candidate encoder before typed validation. Serialized
|
||||
validators and attempt debug use that candidate representation, which carries
|
||||
the codec media type and schema identity but is never checkpointed or passed
|
||||
downstream. Only a validator-approved value is encoded through the strict final
|
||||
codec and made eligible for a checkpoint or stage output.
|
||||
|
||||
Checkpoint identity, physical layout, reuse behavior, and debug artifact
|
||||
handling are operator contracts in [Operations](../operations.md). Serialization
|
||||
and recorder implementation are inventoried in
|
||||
[Internal Overview](overview.md#run-state-components).
|
||||
|
||||
## Results And Failures
|
||||
|
||||
The runner owns manifest assembly and handoff summaries but not the durable JSON
|
||||
schema. It records resolved module and lane provenance, validator chains,
|
||||
source/reference identities, selected LLM profiles, normalized and rejected
|
||||
summaries, status, and timing. Serialized artifact content remains outside the manifest.
|
||||
Module metadata providers may add non-secret singleton or lane-scoped metadata.
|
||||
|
||||
Execution errors include stage, module, lane, or validator context. Once a
|
||||
manifest exists, a failing run returns it with failed status and completion
|
||||
time. Successful status reflects whether any result was rejected. The
|
||||
durable manifest and logical file schemas are defined in the
|
||||
[JSON output contract](../integrations/json-output.md).
|
||||
|
||||
On a framework failure, the runner cancels its derived context, stops submitting
|
||||
new extract work, drains started tasks, and skips the output encoder. Parent
|
||||
cancellation takes precedence. Otherwise context-cancellation fallout is
|
||||
discarded when a substantive error exists, and the primary error is selected by
|
||||
stage, resolved lane, and source chunk rather than completion time.
|
||||
|
||||
## Tests To Inspect
|
||||
|
||||
- `internal/core/config/effective_config_test.go`: config-to-resolution boundary.
|
||||
- `internal/framework/pipeline/profile_test.go`: selection, defaults,
|
||||
capabilities, validator chains, and digest behavior.
|
||||
- `internal/framework/pipeline/artifact_codec_registry_test.go`: typed codec
|
||||
metadata, registration, erasure safety, strict decoding, and cloning.
|
||||
- `internal/framework/pipeline/typed_resolution_test.go`: heterogeneous typed
|
||||
lane resolution and preparation, target-specific validators,
|
||||
incompatibilities, ordering, and schema-sensitive pipeline identity.
|
||||
- `internal/framework/pipeline/runner_concurrency_test.go`: bounded dispatch and
|
||||
continuations, reverse completion, stable errors, rejection, cancellation,
|
||||
retries, and independent provider-call limits.
|
||||
- `internal/framework/pipeline/preparation_test.go`: option validation,
|
||||
construction order, dependency failures, and the before-source-work boundary.
|
||||
- `internal/framework/pipeline/references_test.go`: target resolution and
|
||||
materialization.
|
||||
- `internal/cli/run_contract_test.go`: production run transitions, retries,
|
||||
rejections, warnings, CLI recomputation controls, debug hooks, and manifests.
|
||||
- `internal/cli/recompute_execution_contract_test.go`: filesystem-backed
|
||||
selective recomputation and accepted-producer recovery.
|
||||
- `internal/cli/production_contract_test.go`: production composition and
|
||||
configuration-resolution smoke coverage.
|
||||
- `internal/cli/example_contract_test.go`: maintained example resolution and
|
||||
execution ownership.
|
||||
- `internal/modules/integration/*_test.go` and
|
||||
`internal/modules/seriatim/input/transcript/runner_test.go`: typed runner
|
||||
composition across concrete module families.
|
||||
- `internal/framework/checkpoint/*_test.go`: checkpoint serialization and reuse
|
||||
collaborators.
|
||||
Run **go test ./internal/framework/pipeline ./internal/cli** after changing a
|
||||
pipeline boundary. Use the more focused tests above while iterating.
|
||||
|
||||
@@ -2,7 +2,8 @@
|
||||
|
||||
This document describes the implementation collaborators behind output, cache,
|
||||
and debug state. User-visible fields belong in [Configuration](../config.md),
|
||||
and layouts and lifecycle belong in [Operations](../operations.md).
|
||||
and physical layout, retention, recovery, reason codes, and cleanup belong in
|
||||
[Operations](../operations.md).
|
||||
|
||||
## Composition
|
||||
|
||||
@@ -12,11 +13,22 @@ constructs cache collaborators, writes logical output files, and reports paths.
|
||||
Pipeline modules receive interfaces and request data, never output, cache, or
|
||||
debug roots.
|
||||
|
||||
The CLI creates no chunk-plan store in bypass mode. It creates a checkpoint
|
||||
recorder only when recording is enabled and a checkpoint loader only for a
|
||||
resume invocation. It allocates debug state only after a safe run identity has
|
||||
been generated and only when debug capture was requested. These choices keep
|
||||
the three state families independently composable.
|
||||
|
||||
## Output And Cache
|
||||
|
||||
The pipeline runner returns logical output files. After validating every
|
||||
logical name, the CLI exclusively creates the run directory beneath the
|
||||
selected output root and performs confined, atomic file writes within it.
|
||||
The runner supplies an accepted chunk map as an optional, defensively owned
|
||||
output-request artifact. The JSON encoder alone decides whether its explicit
|
||||
option writes the map and optional index descriptor; neither the map payload
|
||||
nor its annotations are copied into the run manifest. The durable fields are
|
||||
owned by the [Accepted Chunk Map contract](../integrations/chunk-map.md).
|
||||
|
||||
`internal/framework/chunkplan` owns source-addressed plan storage, validation,
|
||||
and atomic publication. Its store is constructed only when the selected mode is
|
||||
@@ -27,7 +39,9 @@ codecs, loader, and recorder. The CLI constructs a recorder whenever checkpoint
|
||||
recording is enabled and constructs a loader only for a `--resume` invocation.
|
||||
Identity incorporates explicit stable semantic fingerprints collected from
|
||||
prepared modules and validators in addition to configuration, input,
|
||||
references, runtime overrides, and LLM profiles.
|
||||
references, runtime overrides, observed LLM profiles, and the LLM runtime's
|
||||
non-secret effective profile-source identity. A profile source change therefore
|
||||
causes a cold miss even when the configured profile ID remains unchanged.
|
||||
The serialized
|
||||
`workspace_schema_version` identifiers are frozen wire-compatibility fields;
|
||||
they do not describe a current public state surface.
|
||||
@@ -57,18 +71,25 @@ records the decision, and stops without executing the producer or consumer.
|
||||
The loader assigns a typed category and reason code at each validation site;
|
||||
diagnostic prose is not classified after the fact. The runner then applies
|
||||
forced-execution policy, validates reusable artifact bytes through the prepared
|
||||
codec, and records the final decision before enforcing a required-predecessor
|
||||
failure. That failure names only the step, lane, and stable reason code. Decision
|
||||
detail passes through one UTF-8-safe bounded sanitizer and contains only
|
||||
allowlisted diagnostic context, never payloads, references, credentials,
|
||||
environment values, or physical paths. Typed categories and codes remain intact
|
||||
through pipeline events and become strings only in manifest and debug-summary
|
||||
JSON. [Operations](../operations.md#resume-and-selective-recompute) is the
|
||||
codec once, returns the canonical hydrated value to the stage, and records the
|
||||
final decision before enforcing a required-predecessor failure. That failure
|
||||
names only the step, lane, and stable reason code. Decision detail is selected
|
||||
from code-owned descriptions by reason code and then UTF-8 normalized and
|
||||
bounded; callers cannot supply arbitrary diagnostic prose. Typed categories and
|
||||
codes remain intact through pipeline events and become strings only in manifest
|
||||
and debug-summary JSON.
|
||||
[Operations](../operations.md#checkpoint-recording-resume-and-recompute) is the
|
||||
canonical operator-facing reason-code reference.
|
||||
|
||||
`internal/core/fileio` provides confined atomic file writes used by state
|
||||
collaborators. The chunk-plan store retains its stronger entry validation.
|
||||
|
||||
The CLI constructs selective-recomputation policy from resolved generated
|
||||
artifact dependencies. It forces the selected step and transitive consumers,
|
||||
while marking unforced producers as required reusable inputs. The runner owns
|
||||
the actual hydration and rejection decisions; the [Operations guide](../operations.md#checkpoint-recording-resume-and-recompute)
|
||||
owns the operator workflow and stable reason-code meanings.
|
||||
|
||||
## Debug Bundles
|
||||
|
||||
`internal/core/debugbundle` allocates an explicitly requested per-run bundle
|
||||
@@ -93,12 +114,27 @@ operation writes the success report, or makes one attempt each to write the
|
||||
failure report and error log. Terminal persistence failures are reported
|
||||
separately and never replace the command's primary error.
|
||||
|
||||
## Invariants To Preserve
|
||||
|
||||
- Modules receive state collaborators and request data, never physical roots.
|
||||
- Output logical paths are validated before a run directory is allocated, and
|
||||
files are atomically written within that directory.
|
||||
- Chunk-plan publication occurs only for accepted plans; bypass does not
|
||||
construct or touch a plan store.
|
||||
- Checkpoint recording and checkpoint loading remain separate collaborators.
|
||||
- Debug state is opt-in, is not cache input, and terminal reporting does not
|
||||
obscure the command's primary failure.
|
||||
|
||||
## Tests To Inspect
|
||||
|
||||
- `internal/cli/run_contract_test.go`: command-owned state allocation,
|
||||
terminalization, and output/report boundaries.
|
||||
- `internal/cli/cache_contract_test.go`: cache-mode precedence, root selection,
|
||||
and resume collaborator construction.
|
||||
- `internal/cli/state_hardening_test.go`: independent roots, reuse, failures,
|
||||
permissions, cleanup, and redaction.
|
||||
- `internal/cli/recompute_policy_test.go`: forced dependents and required
|
||||
reusable predecessors for selective recomputation.
|
||||
- `internal/cli/recompute_execution_contract_test.go`: selective recomputation,
|
||||
filesystem recovery, deterministic decisions, and failed predecessor state.
|
||||
- `internal/cli/production_contract_test.go`: production composition and
|
||||
|
||||
@@ -1,287 +1,291 @@
|
||||
# Operations
|
||||
|
||||
This is the canonical guide to operating Notarius filesystem state. Command
|
||||
syntax is in the [CLI reference](cli.md); field definitions and precedence are
|
||||
in [Configuration](config.md).
|
||||
This is the canonical guide for operating Notarius runtime state. The
|
||||
[CLI reference](cli.md) owns command syntax and exit statuses, while
|
||||
[Configuration](config.md) owns fields, defaults, and precedence. Maintainers
|
||||
who need implementation mechanics should read [Run State Internals](internal/state.md).
|
||||
|
||||
## State Model
|
||||
## State Surfaces
|
||||
|
||||
Notarius uses three independent filesystem surfaces:
|
||||
Each run can use independent roots with different retention and access-control
|
||||
needs.
|
||||
|
||||
- output is durable user data;
|
||||
- cache is reconstructible chunk-plan and checkpoint state; and
|
||||
- debug is explicitly requested inspection data.
|
||||
| Surface | Purpose | Created when | Retention |
|
||||
| --- | --- | --- | --- |
|
||||
| Output | Durable user-facing result bundle | A pipeline completes and returns logical output files | Keep until consumers no longer need it. |
|
||||
| Chunk-plan cache | Reconstructible source-addressed plan | The configured cache mode permits cache I/O | Keep while reuse is useful. |
|
||||
| Checkpoint cache | Reconstructible execution and recovery state | Checkpoint recording is enabled | Keep only while recovery or reuse is useful. |
|
||||
| Debug bundle | Explicit diagnostic record | A run requests debug collection | Keep only under an intentional sensitive-data retention policy. |
|
||||
|
||||
Choose separate roots and access controls for each surface. A normal run writes
|
||||
durable output, may use the chunk-plan cache, and records checkpoints when
|
||||
`cache.checkpoints.enabled` is true. It does not create debug state unless its
|
||||
invocation includes `--debug`.
|
||||
Output, cache, and debug roots are never merged or cleaned automatically. Use
|
||||
separate locations and permissions for operators or services that must not
|
||||
share application data.
|
||||
|
||||
## Output
|
||||
## Roots And Permissions
|
||||
|
||||
Durable logical files are written under:
|
||||
The configured output and debug directories are exact roots. An empty cache
|
||||
directory selects a per-user root:
|
||||
|
||||
```text
|
||||
~~~
|
||||
<os.UserCacheDir>/notarius/chunk-plans
|
||||
<os.UserCacheDir>/notarius/checkpoints
|
||||
~~~
|
||||
|
||||
The field definitions and configuration examples are in [Configuration](config.md).
|
||||
On supported Unix systems, output directories and files are created with
|
||||
requested modes **0755** and **0644**. Chunk-plan, checkpoint, and debug
|
||||
directories and files use **0700** and **0600**. The operating system's umask
|
||||
may impose stricter output modes. Cache and debug roots may contain sensitive
|
||||
source-derived data, so provision them for one trusted account or service. An
|
||||
output bundle can also contain source content when its JSON output enables
|
||||
evidence publication. Apply an appropriate umask and output-root access policy
|
||||
before enabling that option; the requested output modes alone may not be
|
||||
suitable for transcript-bearing bundles.
|
||||
|
||||
## PromptKit Profile Deployment
|
||||
|
||||
Profile deployment has four distinct layers:
|
||||
|
||||
| Layer | Owner | Operational role |
|
||||
| --- | --- | --- |
|
||||
| Prompts and schemas | Notarius module families | Embedded request and structured-output definitions. They are not deployment profile files. |
|
||||
| Fallback profiles | Notarius module families | Embedded application defaults, including D&D's `dnd-extraction` profile. |
|
||||
| Built-in profiles | PromptKit | Upstream catalog entries available when no higher-precedence source defines an ID. |
|
||||
| Operator profiles | Deployment filesystem | Complete environment-specific definitions selected by `promptkit.profile_file` or `promptkit.profile_dir`. |
|
||||
|
||||
The maintained D&D pipeline uses the workload ID `dnd-extraction`. The
|
||||
embedded fallback makes that ID usable without an operator file. Production,
|
||||
development, and local deployments can each install a different complete
|
||||
definition for the same ID, retaining the pipeline while choosing their own
|
||||
model, backend, timeout, or reasoning policy. An operator definition wins over
|
||||
the fallback; it is not merged with it. The configuration field and full
|
||||
precedence rules are owned by [Configuration](config.md#promptkit-profiles).
|
||||
|
||||
Use a profile source owned by the service account, keep it readable only by
|
||||
the intended operator, and supply provider credentials through the service
|
||||
environment—not in the Notarius configuration or profile YAML. The maintained
|
||||
[operator profile](../examples/profiles/dnd-extraction.yml) is secret-free and
|
||||
can be copied as a format starting point. Validate a deployment without a
|
||||
provider call or credentials:
|
||||
|
||||
~~~sh
|
||||
notarius config validate --config /etc/notarius/config.yml --pipeline dnd-session
|
||||
~~~
|
||||
|
||||
Profile paths are currently resolved from the process working directory, not
|
||||
from the configuration file. The complete example's
|
||||
`./examples/profiles/dnd-extraction.yml` path is valid for a repository-root
|
||||
invocation only. Use absolute paths such as
|
||||
`/etc/notarius/profiles/dnd-extraction.yml` for services and containers.
|
||||
|
||||
## Run Lifecycle
|
||||
|
||||
Use the [run command](cli.md#run) to start a pipeline. A valid invocation loads
|
||||
and resolves configuration before module preparation and source parsing. It
|
||||
then performs any permitted cache lookup, executes the pipeline, and publishes
|
||||
logical output files only after a successful runner result.
|
||||
|
||||
On success, the command reports the output bundle path. A warning-bearing run
|
||||
still succeeds and reports its warning count on standard error. Errors and
|
||||
their exit classes are defined in the [CLI reference](cli.md#output-streams-and-exit-statuses).
|
||||
|
||||
## Output Bundles
|
||||
|
||||
Each successful run receives a generated safe run identifier and writes beneath:
|
||||
|
||||
~~~
|
||||
<output-root>/<run-id>/
|
||||
```
|
||||
~~~
|
||||
|
||||
The CLI generates one run ID in the form
|
||||
`run-<started-at-unix-nanoseconds>-<32-lowercase-hex-characters>` and uses it
|
||||
for output, manifests, and any requested debug bundle. It validates every
|
||||
logical output name before exclusively creating the run directory. If that
|
||||
directory already exists, the invocation fails without changing it.
|
||||
The [JSON output contract](integrations/json-output.md) owns the logical files
|
||||
and their schemas. Before creating the run directory, Notarius validates every
|
||||
logical output path. It refuses an existing run directory without changing it.
|
||||
Files are written atomically; if a later write fails, the newly created partial
|
||||
run directory remains for inspection and is never removed automatically.
|
||||
|
||||
Each output file is written atomically. A later file-write failure leaves the
|
||||
newly allocated partial run directory in place for inspection; Notarius never
|
||||
automatically removes output. The
|
||||
[JSON output contract](integrations/json-output.md) owns the logical file
|
||||
names, schemas, and media types inside a run directory.
|
||||
|
||||
Remove an output run directory only after its consumer data is no longer
|
||||
needed. This is data deletion, not cache cleanup.
|
||||
|
||||
## Ordered D&D Workflow
|
||||
|
||||
The maintained [NPC-grounded configuration](../examples/dnd-npc-grounded.config.yml)
|
||||
contains one pipeline with two ordered steps. The first step extracts and
|
||||
normalizes NPCs. Only after that lane reaches an accepted terminal result does
|
||||
the second step begin; its generated NPC reference is supplied in memory to
|
||||
spell extraction, combat extraction, and combat normalization.
|
||||
|
||||
```sh
|
||||
go run ./cmd/notarius run dnd-npc-grounded \
|
||||
--config examples/dnd-npc-grounded.config.yml \
|
||||
--input examples/seriatim-minimal-transcript.json \
|
||||
--output-dir ./npc-grounded-output
|
||||
```
|
||||
|
||||
The NPC artifact grounds canonical names and aliases, not spell or combat
|
||||
evidence. Current-transcript source ranges remain the only event evidence. The
|
||||
manifest records generated-reference identity and bounded producer provenance;
|
||||
it does not record generated payload content, and no generated content is
|
||||
exposed through a filesystem path. The same producer artifact may fan out to
|
||||
compatible consumers, while a missing or rejected producer prevents the later
|
||||
step from starting.
|
||||
|
||||
Standalone module configurations continue to support external NPC files when a
|
||||
workflow intentionally crosses a process or session boundary. Those files are
|
||||
validated against the consumer slot and must be protected as sensitive
|
||||
campaign data. They are not part of the maintained ordered handoff workflow.
|
||||
Treat an output bundle as durable user data. Do not use cache-cleanup policy to
|
||||
remove it. An optional accepted chunk map is also durable output and can carry
|
||||
source- or model-derived annotations; its content and compatibility contract
|
||||
are defined in [Accepted Chunk Map](integrations/chunk-map.md). An optional
|
||||
[evidence context](integrations/evidence-context.md) contains source-unit text
|
||||
and metadata. It is not a cache or debug artifact: retain it with the output
|
||||
bundle only for as long as consumers need it, and apply source-content access
|
||||
controls to the entire bundle. Selected lanes may collectively cite most of a
|
||||
transcript, so a broad allowlist can make the evidence artifact nearly as
|
||||
sensitive and large as the source itself.
|
||||
|
||||
## Chunk-Plan Cache
|
||||
|
||||
Chunk plans are stored at:
|
||||
Chunk plans live beneath the selected chunk-plan root:
|
||||
|
||||
```text
|
||||
~~~
|
||||
<chunk-plan-root>/<source-sha256-hex>/plan.json
|
||||
```
|
||||
~~~
|
||||
|
||||
`auto` reuses a complete valid plan or regenerates missing or invalid state.
|
||||
`refresh` regenerates and atomically replaces a plan after chunk validation.
|
||||
`bypass` performs no plan-cache I/O and does not resolve or create the root.
|
||||
Plan selection is source-addressed and independent of checkpoint and debug
|
||||
roots.
|
||||
One validated canonical plan is active for each source digest. The plan stores
|
||||
boundaries and provenance, not a second copy of the entire source. This
|
||||
source-addressed policy is recorded in [ADR-0005](adr/0005-cache-canonical-chunk-plans-by-source.md).
|
||||
|
||||
When its directory is empty in configuration, the root is
|
||||
`<os.UserCacheDir>/notarius/chunk-plans`. A configured directory is the exact
|
||||
root; no suffix is appended. Directories and files created by the store use
|
||||
`0700` and `0600` permissions on supported Unix systems. The configured root
|
||||
is a trust boundary: do not share it among mutually untrusted users.
|
||||
The configured cache mode controls one invocation:
|
||||
|
||||
Remove an exact digest directory or the configured root only when accepting the
|
||||
cost of recomputing plans and any chunk-stage work. Cache publication is atomic;
|
||||
there is no history, locking, garbage collection, or rollback facility.
|
||||
- **auto** looks for a valid active plan. Missing or invalid state causes a new
|
||||
plan to be generated; an accepted new plan is atomically published.
|
||||
- **refresh** skips lookup, generates a plan with the configured chunker, and
|
||||
atomically replaces the active plan after it is accepted.
|
||||
- **bypass** performs no chunk-plan cache I/O. It does not resolve or create a
|
||||
chunk-plan root.
|
||||
|
||||
For a Linux service account, provision a dedicated restrictive root such as:
|
||||
A reused plan is still materialized and validated against the current source.
|
||||
If a prior plan no longer gives acceptable results, use a refresh run rather
|
||||
than editing cache files. Deleting a plan is recoverable but can repeat costly
|
||||
chunking work.
|
||||
|
||||
```yaml
|
||||
cache:
|
||||
chunk_plans:
|
||||
directory: /var/cache/notarius/chunk-plans
|
||||
```
|
||||
## Checkpoint Recording, Resume, And Recompute
|
||||
|
||||
## Checkpoint Cache
|
||||
Checkpoint recording is an explicit configuration choice and is disabled by
|
||||
default. When enabled, each run records stage transitions and the state needed
|
||||
for compatible recovery. A run records checkpoints even when it does not ask
|
||||
to reuse them. Checkpoint payloads can contain source-derived and intermediate
|
||||
application data, so treat the entire root as sensitive.
|
||||
|
||||
Checkpoint recording is controlled by `cache.checkpoints.enabled`, which
|
||||
defaults to `false`. When enabled, every run records running, succeeded, and
|
||||
failed transitions and reusable validator-approved results. Successful,
|
||||
rejected, and failed runs may therefore all leave checkpoint state. The
|
||||
`--resume` flag additionally loads compatible completed work before executing
|
||||
missing or incompatible stages. Without `--resume`, a recording-enabled run
|
||||
never loads checkpoints. Using `--resume` while recording is disabled is an
|
||||
error.
|
||||
Checkpoint loading is separate: [**--resume**](cli.md#run) asks a run to reuse
|
||||
compatible recorded work. A resume request fails when checkpoint recording is
|
||||
disabled. Without **--resume**, a recording-enabled run executes normally and
|
||||
does not load checkpoint state. Compatibility includes the resolved pipeline,
|
||||
input, selected lanes, runtime overrides, reference provenance, LLM-profile
|
||||
provenance, the effective PromptKit profile-source fingerprint, and
|
||||
prepared-component fingerprints. When a local PromptKit backend is configured,
|
||||
compatibility also includes a non-secret fingerprint of its endpoint. Changing
|
||||
profile content or the local endpoint causes a cold miss; changing only the
|
||||
local concurrency limit does not. A changed identity produces a cold miss;
|
||||
Notarius does not migrate, rewrite, or delete older checkpoint directories.
|
||||
Reasoning-effort inheritance, replacement, and explicit clearing are distinct
|
||||
runtime identities, so checkpoints created under one state are not reused by
|
||||
either of the others.
|
||||
|
||||
Checkpoints use the selected root and the existing identity hierarchy:
|
||||
Checkpoint state is confined below an identity-specific path:
|
||||
|
||||
```text
|
||||
<checkpoint-root>/<pipeline-id>/<input-key>-<source-or-input-digest>/<pipeline-digest>/<identity-digest>/...
|
||||
```
|
||||
~~~
|
||||
<checkpoint-root>/<pipeline-id>/<input-key>-<source-or-input-digest-prefix>/<pipeline-digest-prefix>/<identity-digest-prefix>/
|
||||
~~~
|
||||
|
||||
The final identity digest includes stable semantic fingerprints explicitly
|
||||
contributed by prepared modules and validators. Adding or changing one of
|
||||
these fingerprints intentionally causes a cold cache miss; old checkpoint
|
||||
directories are left in place and are never migrated or deleted automatically.
|
||||
### Selective Recompute
|
||||
|
||||
An empty configured directory selects
|
||||
`<os.UserCacheDir>/notarius/checkpoints`. The root is exact when configured.
|
||||
Created directories and files use `0700` and `0600` permissions on supported
|
||||
Unix systems.
|
||||
[**--recompute-step**](cli.md#run) requires both **--resume** and enabled
|
||||
checkpoint recording. It forces the selected ordered step and every lane that
|
||||
depends on it through generated artifact references. Unrelated lanes remain
|
||||
eligible for reuse.
|
||||
|
||||
Checkpoint payloads can contain source text, intermediate artifacts, metadata,
|
||||
warnings, and content digests. Treat them as sensitive derived application
|
||||
data. Compatible files from a former checkpoint root remain reusable when
|
||||
`cache.checkpoints.directory` names that exact existing root. They are not
|
||||
moved, migrated, or deleted automatically. The frozen serialized identifier
|
||||
`workspace_schema_version` remains part of checkpoint compatibility; it is not
|
||||
a configuration setting.
|
||||
For an earlier producer required by a forced consumer, Notarius requires a
|
||||
compatible accepted normalized artifact. It validates that artifact before
|
||||
hydrating it and does not silently rerun the producer. If that state is
|
||||
missing, rejected, corrupt, non-canonical, or incompatible, the run stops
|
||||
before its dependent starts. Rerun the required producer deliberately instead
|
||||
of copying or editing checkpoint files.
|
||||
|
||||
For a Linux service account, independently provision:
|
||||
## Checkpoint Decisions And Recovery
|
||||
|
||||
```yaml
|
||||
cache:
|
||||
checkpoints:
|
||||
enabled: true
|
||||
directory: /var/cache/notarius/checkpoints
|
||||
```
|
||||
Checkpoint events classify work as **executed**, **reused**,
|
||||
**forced_recompute**, or **dependency_invalidated**. Their stable reason codes
|
||||
are written to run diagnostics and provenance. Use the code, not a copied
|
||||
error message, to decide what to repair.
|
||||
|
||||
Remove an exact checkpoint identity directory or the configured root only when
|
||||
recomputation is acceptable.
|
||||
|
||||
### Resume And Selective Recompute
|
||||
|
||||
`--resume` loads compatible accepted work only when checkpoint recording is
|
||||
enabled. A normal resumed run may reuse source, extract, merge, and normalize
|
||||
checkpoints independently and may recompute a stage after a cache miss.
|
||||
Generated references add a dependency fingerprint
|
||||
for the producer's artifact kind, schema identity, media type, canonical
|
||||
content digest, and size. If that fingerprint changes or the producer is
|
||||
missing, dependent checkpoints are invalidated; unrelated work remains eligible
|
||||
for reuse.
|
||||
|
||||
`--recompute-step <step-id>` requires both `--resume` and
|
||||
`cache.checkpoints.enabled: true`. It forces the named step and all transitive
|
||||
dependents to execute, while compatible predecessors and unrelated lanes remain
|
||||
reusable. The ID may be an explicit configured step or `default` for an
|
||||
implicit single-step pipeline. It cannot be combined with `--only`, and it does
|
||||
not change the persistent identity of otherwise identical checkpoints.
|
||||
Decisions are bounded and categorized as `reused`, `executed`,
|
||||
`forced_recompute`, or `dependency_invalidated`.
|
||||
|
||||
For an unselected producer required by a recomputed step, Notarius loads the
|
||||
accepted normalized artifact directly. Valid normalize state is sufficient even
|
||||
when that producer's extract or merge checkpoint is missing or corrupt. The
|
||||
normalize manifest must be successful and match workspace schema v3, the exact
|
||||
current invocation identity, step, lane, and normalizer; its payload digest and
|
||||
canonical codec representation must also validate. A forced producer bypasses
|
||||
this lookup and executes.
|
||||
|
||||
If a required predecessor's accepted normalized artifact is missing, rejected,
|
||||
corrupt, non-canonical, or incompatible, the run fails before the dependent
|
||||
step starts. It does not fall back to rerunning that predecessor. The failure
|
||||
manifest retains completed upstream outcomes and dependency context but not
|
||||
generated reference content. For diagnosis, first check the producer step and
|
||||
lane in the manifest, then inspect checkpoint decision categories and reason
|
||||
codes. Rerun the producer explicitly rather than copying an artifact into the
|
||||
checkpoint root.
|
||||
|
||||
The decision that caused a required-predecessor failure is retained before the
|
||||
run returns, and the CLI error identifies its step, lane, and reason code.
|
||||
|
||||
Checkpoint reason codes are stable diagnostic identifiers:
|
||||
|
||||
| Reason code | Operator meaning |
|
||||
| Reason code | Recovery meaning |
|
||||
| --- | --- |
|
||||
| `loading_disabled` | This invocation did not enable checkpoint loading. |
|
||||
| `checkpoint_missing` | The requested checkpoint file does not exist. |
|
||||
| `checkpoint_path_invalid` | The requested checkpoint location failed confinement validation. |
|
||||
| `checkpoint_read_failed` | An existing checkpoint could not be read. |
|
||||
| `checkpoint_decode_failed` | Checkpoint JSON could not be decoded. |
|
||||
| `workspace_schema_incompatible` | The stored workspace schema is not supported by this build. |
|
||||
| `identity_mismatch` | The stored invocation identity differs from the current invocation. |
|
||||
| `stage_mismatch`, `step_mismatch`, `lane_mismatch`, `module_mismatch` | Stored scope does not match the requested pipeline scope. |
|
||||
| `status_not_reusable` | The stored operation did not finish in a reusable status. |
|
||||
| `dependency_mismatch` | Stored dependencies differ; the category is `dependency_invalidated`. |
|
||||
| `artifact_payload_invalid` | Stored artifact payload structure or encoding is invalid. |
|
||||
| `artifact_digest_mismatch` | Stored artifact bytes do not match their recorded digest. |
|
||||
| `artifact_codec_incompatible` | Stored artifact identity is incomplete or incompatible with the codec contract. |
|
||||
| `artifact_not_canonical` | The codec can decode the artifact, but its bytes are not canonical. |
|
||||
| `checkpoint_reused` | The stored checkpoint passed validation and was reused. |
|
||||
| `accepted_artifact_reused` | A required producer's accepted normalized artifact was canonically validated and hydrated. |
|
||||
| `recompute_step` | Selective recomputation forced execution of this lane. |
|
||||
| **loading_disabled** | This invocation did not permit checkpoint loading. |
|
||||
| **checkpoint_missing**, **checkpoint_path_invalid**, **checkpoint_read_failed**, **checkpoint_decode_failed** | The stored checkpoint could not be located or read safely; normal resume work can execute again. |
|
||||
| **workspace_schema_incompatible**, **identity_mismatch**, **stage_mismatch**, **step_mismatch**, **lane_mismatch**, **module_mismatch** | Stored state belongs to a different compatible scope or identity; allow a fresh run to create new state. |
|
||||
| **status_not_reusable** | The recorded operation did not end in reusable state. |
|
||||
| **dependency_mismatch** | A dependency changed; dependent work is invalidated rather than reused. |
|
||||
| **artifact_payload_invalid**, **artifact_digest_mismatch**, **artifact_codec_incompatible**, **artifact_not_canonical** | A stored artifact cannot safely be hydrated; rerun the producer instead of modifying the cache. |
|
||||
| **checkpoint_reused** | A normal checkpoint passed compatibility checks. |
|
||||
| **accepted_artifact_reused** | A required predecessor's accepted normalized artifact was safely hydrated. |
|
||||
| **recompute_step** | Selective recomputation deliberately forced this work. |
|
||||
|
||||
Decision detail is bounded explanatory text, not a data-recovery channel. It
|
||||
never contains checkpoint paths, artifact or reference content, source content,
|
||||
credentials, or environment values.
|
||||
Reason detail is bounded code-owned text. It is diagnostic information, not a
|
||||
path-discovery or data-recovery mechanism, and does not contain checkpoint,
|
||||
source, reference, credential, or environment content.
|
||||
|
||||
## Debug Bundles
|
||||
|
||||
Only `notarius run --debug` enables debug collection. The selected root contains
|
||||
one retained bundle per invocation:
|
||||
Only a [debug-enabled run](cli.md#run) creates a bundle:
|
||||
|
||||
```text
|
||||
~~~
|
||||
<debug-root>/<run-id>/
|
||||
summary/
|
||||
trace/
|
||||
```
|
||||
~~~
|
||||
|
||||
`summary/` contains redacted invocation, effective-configuration, resolved
|
||||
pipeline and reference provenance, checkpoint and chunk-plan decisions, run
|
||||
manifest, warnings, report, and any available error text. It excludes raw
|
||||
source, references, annotations, prompts, model responses, credentials, and
|
||||
malformed cache bytes.
|
||||
The summary contains redacted invocation and resolution information plus run,
|
||||
warning, checkpoint, chunk-plan, and terminal reporting artifacts. The trace
|
||||
contains allowlisted application diagnostic records and can include source or
|
||||
derived application data. Neither surface is a cache input. Do not treat a
|
||||
debug bundle as safe to share merely because its configuration summary is
|
||||
redacted. Invocation metadata omits reasoning effort when it is inherited,
|
||||
records the replacement value when one is supplied, and records an empty value
|
||||
when inherited reasoning was explicitly cleared.
|
||||
|
||||
`trace/` contains application-owned execution detail, including source and
|
||||
stage material, plans, chunks, validator attempts, prompts, model responses,
|
||||
timing, and serialized artifacts. It may retain application data omitted from
|
||||
output. Credentials, credential-shaped values, sensitive metadata, unrelated
|
||||
environment values, and unrelated filesystem content are not captured.
|
||||
|
||||
Bundles inherit the sensitivity of the application data they capture. Their
|
||||
additional risk comes from copying and aggregating that data, so restrict
|
||||
access, avoid shared roots between untrusted users, and define retention outside
|
||||
Notarius. Created bundle directories use `0700` and files use `0600` on
|
||||
supported Unix systems.
|
||||
|
||||
Notarius never automatically deletes a requested bundle. If allocation
|
||||
succeeds, its path is reported on success and failure. A requested summary or
|
||||
trace write failure makes the command fail, preserving whatever bundle data was
|
||||
already written for inspection. Every allocated bundle makes one best-effort
|
||||
attempt to record a terminal `run-report.json`.
|
||||
|
||||
## Failures And Warnings
|
||||
|
||||
Failures before debug allocation are reported on stderr without a bundle.
|
||||
Failures after allocation report the bundle path on stderr and make independent
|
||||
attempts to write a failure `run-report.json` and `error.log`. The report retains
|
||||
the paths and pipeline outcome fields known at the failure point. If either
|
||||
terminal write fails, the original command error remains first on stderr,
|
||||
followed by the persistence error and bundle path. An output-write failure
|
||||
leaves the allocated bundle in place. A successful run with warnings exits `0`,
|
||||
reports a warning count on stderr, and records warnings in durable output and
|
||||
any requested debug summary.
|
||||
Notarius never creates debug state without an explicit request and never
|
||||
automatically deletes a requested bundle. If allocation succeeds, the command
|
||||
reports its path on both success and later failure. A summary, trace, or
|
||||
terminal-report persistence failure fails the command while preserving any
|
||||
already-written diagnostic data for inspection.
|
||||
|
||||
## Cleanup
|
||||
|
||||
Use exact paths for manual cleanup. Examples:
|
||||
Cleanup is manual and destructive. First inspect the exact leaf directory,
|
||||
then remove only that leaf; do not use a glob or a parent root as the target.
|
||||
|
||||
```sh
|
||||
rm -rf ./notarius-output/run-1721300000000000000-0123456789abcdef0123456789abcdef
|
||||
rm -rf /var/cache/notarius/chunk-plans/0123abcd
|
||||
rm -rf /var/cache/notarius/checkpoints/pipeline/input-0123/pipeline-4567/identity-89ab
|
||||
rm -rf ./notarius-debug/run-1721300000000000000-0123456789abcdef0123456789abcdef
|
||||
```
|
||||
~~~
|
||||
rm -rf -- /srv/notarius/output/run-1721300000000000000-0123456789abcdef0123456789abcdef
|
||||
rm -rf -- /srv/notarius/chunk-plans/0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef
|
||||
rm -rf -- /srv/notarius/checkpoints/example/seriatim-0123456789abcdef/0123456789abcdef/0123456789abcdef
|
||||
rm -rf -- /srv/notarius/debug/run-1721300000000000000-0123456789abcdef0123456789abcdef
|
||||
~~~
|
||||
|
||||
Avoid broad recursive cleanup against a parent root unless it is an explicit
|
||||
operator policy. Output deletion is permanent user-data loss. Cache deletion is
|
||||
recoverable but can repeat expensive work. Debug deletion removes troubleshooting
|
||||
evidence and any retained application-data copy.
|
||||
Deleting output permanently removes user data. Deleting chunk plans or
|
||||
checkpoints is recoverable but may repeat expensive provider or pipeline work.
|
||||
Deleting a debug bundle removes troubleshooting evidence and a retained copy of
|
||||
application data. Notarius has no cache garbage collector, rollback operation,
|
||||
or automatic cleanup command.
|
||||
|
||||
## Operational Limits
|
||||
|
||||
Provider retries and timeouts are handled by Scriptorium according to the
|
||||
selected execution profile. Pipeline module retry settings are defined in
|
||||
[Configuration](config.md#module-bindings). Extract worker concurrency and
|
||||
actual provider-call concurrency are separate limits; their fields and
|
||||
validation are defined in [Configuration](config.md#concurrency). Notarius
|
||||
writes local files only; remote storage and archive management are outside the
|
||||
implemented CLI.
|
||||
Provider execution settings and the generation timeout come from the selected
|
||||
PromptKit profile. The invocation-only **--reasoning-effort** and
|
||||
**--clear-reasoning-effort** controls may replace or clear that profile setting
|
||||
for all LLM-backed calls in one run without changing the profile. PromptKit
|
||||
v0.5.0 does not add a provider retry loop. Notarius binding retries rerun the
|
||||
complete module operation and validation chain as defined by
|
||||
[module bindings](config.md#module-bindings-and-validators).
|
||||
|
||||
Timeouts are layered. Caller cancellation is the outer authority. A positive
|
||||
effective generation timeout adds an inner request deadline, while zero
|
||||
disables only that generation deadline. The HTTP client timeout remains a
|
||||
transport-wide cap. Notarius does not add another timeout around PromptKit.
|
||||
The pinned upstream boundary and profile-format links are in
|
||||
[PromptKit Integration](integrations/pkg-promptkit.md).
|
||||
|
||||
Concurrency has two independent layers. Notarius **total_llm** is the
|
||||
application-wide provider-call limit shared by all backends, modules, retries,
|
||||
and validators. PromptKit may impose a narrower admission limit for the
|
||||
selected backend. The effective active-generation bound is the intersection of
|
||||
both limits and can therefore be lower than **total_llm**. Built-in OpenRouter
|
||||
profiles use PromptKit's upstream backend limit; endpoint-only profiles have no
|
||||
PromptKit backend limit and remain bounded by Notarius. For the configured
|
||||
local backend, a zero **concurrency_limit** leaves only the Notarius scheduler
|
||||
as a call limit. A positive value makes the effective active local-generation
|
||||
bound the smaller of **total_llm** and that local limit.
|
||||
|
||||
For a positive local limit, PromptKit owns its default waiting capacity and
|
||||
admission behavior. When a PromptKit backend has admitted all active and queued
|
||||
work, a new call fails as capacity exhaustion before generation. The adapter
|
||||
maps that failure to Notarius's existing provider-neutral capacity error and
|
||||
does not retry it. The calling stage's configured retry policy applies
|
||||
normally, and the run fails if those attempts are exhausted. Caller
|
||||
cancellation remains authoritative. Configuration contracts are documented
|
||||
under [PromptKit profiles](config.md#promptkit-profiles) and
|
||||
[concurrency](config.md#concurrency-output-cache-and-debug). Extract-worker
|
||||
limits and actual provider-call limits are independent. Notarius writes local
|
||||
filesystem state only; remote storage, archival, and retention automation are
|
||||
outside the implemented CLI.
|
||||
|
||||
@@ -79,6 +79,14 @@ Pipeline resolution requires a compatible codec and matching kind-specific
|
||||
variants before a typed lane can be accepted. Framework-owned erasure remains
|
||||
private and must report type incompatibility as an error rather than a panic.
|
||||
|
||||
An artifact kind may additionally provide a typed evidence projection that
|
||||
copies its direct generic source references. Preparation proves that projection
|
||||
matches the artifact codec's exact Go type before retaining it for an output
|
||||
policy. The runner reconstructs evidence only from accepted serialized
|
||||
normalized artifacts, and the output boundary owns any resulting publication.
|
||||
Generic framework code never infers evidence by inspecting domain JSON or
|
||||
depends on domain artifact types.
|
||||
|
||||
Auxiliary references provide context or disambiguation. They are not source
|
||||
evidence and must not be converted into source references.
|
||||
|
||||
@@ -169,7 +177,8 @@ individual modules.
|
||||
The application-wide LLM scheduler bounds actual provider calls independently
|
||||
of framework worker limits. Every LLM-backed module, retry, and validator uses
|
||||
the single injected scheduled client, including work performed by overlapping
|
||||
lanes.
|
||||
lanes. Provider runtime adapters may enforce a narrower backend-specific limit
|
||||
beneath this mandatory application-wide scheduler.
|
||||
|
||||
## Configuration And Provenance
|
||||
|
||||
|
||||
@@ -7,6 +7,37 @@ not as committed release dates.
|
||||
|
||||
## Near-Term D&D Pipeline
|
||||
|
||||
### Combat Enemy Ledger
|
||||
|
||||
- Add a D&D artifact that identifies enemies faced during combat and supports
|
||||
an end-of-session encounter ledger.
|
||||
- Track each enemy's observed state using a small controlled vocabulary such as
|
||||
`active`, `killed`, `fled`, `captured`, or `incapacitated`, while preserving
|
||||
an explicit unresolved state when the transcript does not establish an
|
||||
outcome.
|
||||
- Preserve the evidence for enemy participation and state changes rather than
|
||||
inferring a terminal outcome from combat ending or an enemy disappearing
|
||||
from the conversation.
|
||||
- Define how repeated mentions, groups of unnamed enemies, summoned or allied
|
||||
creatures, and the same enemy appearing in multiple combats affect identity
|
||||
and ledger entries.
|
||||
- Evaluate whether the ledger should be extracted directly, derived from
|
||||
combat-turn artifacts, or use a sequential pipeline that consumes combat
|
||||
turns and the normalized NPC registry as grounding references.
|
||||
|
||||
### Location Extraction
|
||||
|
||||
- Add a D&D artifact for locations visited by the party or otherwise mentioned
|
||||
in the transcript.
|
||||
- Distinguish observed visits from references, plans, recalled places, and
|
||||
uncertain or inferred locations so a mention alone is not reported as a
|
||||
visit.
|
||||
- Preserve transcript evidence for each visit or mention and reconcile aliases,
|
||||
nested places, and repeated appearances without collapsing distinct
|
||||
locations that share a generic name.
|
||||
- Define how the location artifact should ground later narrative reports and
|
||||
whether future event artifacts should retain canonical location identities.
|
||||
|
||||
### Evaluate Spell Extraction And Normalization
|
||||
|
||||
- Evaluate ordinary extraction retries and the completed normalization path
|
||||
@@ -16,28 +47,76 @@ not as committed release dates.
|
||||
validator, and normalizer development. Treat model-quality review as an
|
||||
iterative human evaluation aid, not a deterministic correctness gate.
|
||||
|
||||
### Expand Sequential D&D Artifacts
|
||||
### Evaluate The Shared D&D Scene Plan
|
||||
|
||||
- Add narrative extraction for scene summaries, party actions, and NPCs
|
||||
encountered when that output proves useful beyond the dedicated NPC artifact.
|
||||
- Use ordered pipeline steps when a later artifact needs an accepted earlier
|
||||
artifact as context. Keep independent lanes in the same step and do not
|
||||
introduce a general DAG or concurrent cross-lane reconciliation model.
|
||||
|
||||
### Improve D&D Scene Classification
|
||||
|
||||
- Extend scene annotations with classifications that downstream extractors can
|
||||
use, including reliable combat and narrative indicators.
|
||||
- Strengthen the scene prompt so every scene containing combat turns is marked
|
||||
as combat, and add validation capable of detecting missing or inconsistent
|
||||
combat classifications.
|
||||
- Allow the combat extractor to no-op for chunks that are not classified as
|
||||
combat, avoiding unnecessary model calls where practical.
|
||||
- Allow a narrative extractor to select the corresponding scene classification
|
||||
rather than processing every chunk indiscriminately.
|
||||
- Reassess whether one shared scene plan provides enough context for NPC,
|
||||
spell, combat, and narrative pipelines after these extractors have real-world
|
||||
usage. Add more complex chunking only in response to demonstrated failures.
|
||||
spell, combat, interaction, and scene-description lanes after real-world use.
|
||||
Add more complex chunking only in response to demonstrated failures.
|
||||
|
||||
## Cross-Cutting LLM Runtime
|
||||
|
||||
### Deterministic Prompt Session Identity
|
||||
|
||||
- Replace the source-document-ID default for prompt sessions with one
|
||||
predictable, procedurally generated session ID for the complete
|
||||
source-processing workload.
|
||||
- Preserve an explicit non-empty `--session-id` as the highest-precedence
|
||||
override. Otherwise, derive the default only from the effective input module
|
||||
identity and the exact raw input bytes.
|
||||
- Use a versioned, bounded representation such as
|
||||
`notarius:v1:<sha256(input-module + NUL + raw-input)>`. The exact encoding
|
||||
must fit PromptKit's session length contract and must not embed source
|
||||
content.
|
||||
- Keep the derived session stable across runs, pipelines, selected lanes,
|
||||
ordered steps, retries, resume, recomputation, LLM profiles, reasoning
|
||||
overrides, and output, debug, or cache settings.
|
||||
- Do not include file-backed references, generated references, reference
|
||||
contents, or the composition of a reference bundle in session derivation.
|
||||
References may change between prompt calls within one pipeline without
|
||||
changing routing affinity.
|
||||
- Resolve the authoritative session before checkpoint construction and use the
|
||||
same value for checkpoint runtime identity, every prompt-facing module,
|
||||
PromptKit's direct session field, the compatibility `session_id` prompt
|
||||
variable, run-manifest metadata, and debug metadata.
|
||||
- Keep routing identity separate from cache and checkpoint content identity.
|
||||
Exact prompt prefixes, reference contents, model settings, and other
|
||||
generation-affecting inputs must continue to participate in their existing
|
||||
hashes and checkpoint fingerprints even though they do not change the
|
||||
session.
|
||||
- Treat the generated value as a provider-visible, stable pseudonymous
|
||||
correlation identifier. Do not introduce an installation-specific HMAC or
|
||||
secret unless a concrete multi-tenant or privacy requirement justifies
|
||||
sacrificing deterministic identity across installations.
|
||||
|
||||
### Raise The Default Application-Wide LLM Limit
|
||||
|
||||
- Raise the default `concurrency.total_llm` value from 1 to 16 so ordinary
|
||||
single-backend runs can use PromptKit's expected OpenRouter capacity and
|
||||
lower-capacity local backends without an unnecessarily narrower Notarius
|
||||
limit.
|
||||
- Keep the Notarius application-wide scheduler mandatory and require
|
||||
`total_llm` to remain a positive integer. Do not make the default unlimited:
|
||||
endpoint-only profiles, an unrestricted local backend, injected clients, and
|
||||
aggregate work across several backends may have no narrower PromptKit limit.
|
||||
- Continue defaulting `concurrency.stage_workers.extract` to the effective
|
||||
`total_llm`, making its default 16 as part of the same change. Preserve an
|
||||
explicit lower extract-worker setting when an operator wants less queued or
|
||||
concurrent extraction work.
|
||||
- Define effective provider concurrency as the intersection of the Notarius
|
||||
application-wide limit, the selected PromptKit backend limit when present,
|
||||
and the work made available by stage execution. A Notarius limit of 16 does
|
||||
not narrow a backend already limited to 16, while a local backend limited to
|
||||
4 remains bounded at 4.
|
||||
- Treat the default as an application-wide safety ceiling across profiles,
|
||||
backends, modules, retries, and validators. A run that intentionally needs
|
||||
the combined capacity of several backends may configure a higher
|
||||
`total_llm` and an appropriate extract-worker count explicitly.
|
||||
- Retain the existing configuration and environment override surfaces. Update
|
||||
canonical configuration, operations, and internal documentation together
|
||||
when the default changes.
|
||||
- Reconsider decoupling the extract-worker default from `total_llm` only after
|
||||
mixed-backend workloads demonstrate a need for a high global emergency
|
||||
ceiling with a lower default work-production rate.
|
||||
|
||||
## Shared Normalization And Quality Work
|
||||
|
||||
|
||||
@@ -1,399 +1,586 @@
|
||||
# Implementation Plan: Ordered Pipeline Follow-Up
|
||||
# PromptKit v0.5 Implementation Plan
|
||||
|
||||
## Status
|
||||
## Objective
|
||||
|
||||
Completed on 2026-07-22. This document retains the implementation sequence for
|
||||
historical context; current contracts are maintained in the canonical CLI,
|
||||
configuration, operations, integration, and internal documentation linked from
|
||||
the [development guide](../development.md).
|
||||
Implement the target state in
|
||||
[PromptKit v0.5 Integration And LLM Profile Policy](promptkit.md). Each numbered
|
||||
stage is intended to be one implementation prompt for a GPT-5.6-Terra coding
|
||||
agent. Complete stages in order and leave the repository buildable, tested, and
|
||||
internally coherent after every stage.
|
||||
|
||||
## Purpose
|
||||
Follow [Architecture](../policy/architecture.md),
|
||||
[Testing Policy](../policy/testing.md), and
|
||||
[Documentation Policy](../policy/documentation.md) throughout. Preserve
|
||||
unrelated user changes. Use `apply_patch` for source and documentation edits,
|
||||
run `gofmt` on changed Go files, and add only tests that protect the behaviors
|
||||
and risks assigned to that stage.
|
||||
|
||||
Record the work that closed the correctness, observability, test, and
|
||||
maintainability gaps in the implemented
|
||||
[Ordered Pipeline Steps](ordered-pipeline-steps.md) feature. That feature
|
||||
roadmap remains the authority for product intent, policy choices, acceptance
|
||||
criteria, and exclusions. This document preserves the ordered,
|
||||
decision-complete implementation sequence for the follow-up work.
|
||||
Do not implement the separate deterministic session-ID or default-concurrency
|
||||
roadmap items as part of this plan. Do not perform paid or credentialed LLM
|
||||
calls.
|
||||
|
||||
## Background
|
||||
## Background Summary
|
||||
|
||||
The original implementation is complete in broad architecture. Pipelines now
|
||||
resolve and prepare one canonical ordered-step model, execute hard barriers
|
||||
between steps, hand accepted normalized artifacts to later operations as
|
||||
canonical generated references, include generated identity in checkpoint and
|
||||
output provenance, support `--recompute-step`, and use generated NPC output at
|
||||
operation time in the D&D spell and combat consumers. Current documentation and
|
||||
maintained examples describe that model.
|
||||
Notarius currently pins PromptKit v0.3.0, calls `Prepare` and then `Run` for one
|
||||
completion, validates profiles through a synthetic prompt, has no application
|
||||
fallback profile source, and accepts LLM profiles only at individual bindings
|
||||
or through the run-wide CLI override. PromptKit v0.5.0 is source-compatible
|
||||
with the current tree; a temporary v0.5.0 module override has already passed
|
||||
`go test ./...`.
|
||||
|
||||
The completed follow-up addressed these narrower gaps:
|
||||
The implementation must nevertheless treat the upstream optional-parameter
|
||||
change as intentional: unset `temperature`, `max_tokens`, and `top_p` remain
|
||||
unset and are omitted from compatible provider requests. Do not restore the old
|
||||
implicit `top_p: 1` default.
|
||||
|
||||
- selective recomputation now hydrates an unselected predecessor from its
|
||||
accepted normalized artifact without requiring its extract and merge state;
|
||||
- the failed checkpoint decision is recorded before a required-predecessor
|
||||
error returns;
|
||||
- checkpoint reason codes are assigned explicitly rather than inferred from
|
||||
human-readable prose;
|
||||
- runner lane and checkpoint orchestration has explicit responsibility seams;
|
||||
and
|
||||
- CLI and resumed-producer acceptance coverage exercises the complete recovery
|
||||
contract.
|
||||
## Stage 1: Upgrade The PromptKit Dependency
|
||||
|
||||
No new product feature is introduced by this plan. Preserve configuration file
|
||||
version 3 and checkpoint workspace schema `notarius.workspace.v3`; the fixes do
|
||||
not require a new persistent format.
|
||||
### Goal
|
||||
|
||||
## Instructions For Every Stage
|
||||
Establish a clean PromptKit v0.5.0 baseline before adopting its new APIs.
|
||||
|
||||
Before changing code, read `docs/development.md`, all documents under
|
||||
`docs/policy/`, and the task-specific internal documents named there. Use the
|
||||
repository's code knowledge graph for code discovery before falling back to
|
||||
text search.
|
||||
### Work
|
||||
|
||||
Implement the stages in order. Each stage must leave the repository formatted,
|
||||
building, and passing its focused tests. Preserve these invariants throughout:
|
||||
- Update `go.mod` and `go.sum` from PromptKit v0.3.0 to v0.5.0 and run
|
||||
`go mod tidy`.
|
||||
- Change the PromptKit built-in profile-catalog marker in
|
||||
`internal/framework/llm/promptkit_profile_fingerprint.go` to identify
|
||||
v0.5.0. This deliberately invalidates LLM checkpoints tied to the prior
|
||||
catalog identity.
|
||||
- Review PromptKit-facing compile errors or test failures against the v0.4.0
|
||||
and v0.5.0 release guides. Do not adopt prepared execution, inspection, or
|
||||
fallback profiles in this stage.
|
||||
- Replace the existing test assertion for one exact built-in fingerprint hash
|
||||
with durable assertions that the fingerprint is deterministic, non-empty,
|
||||
non-secret, and changes when a semantic profile source changes. Do not add a
|
||||
new version-constant or exact-hash change detector.
|
||||
- Update `docs/integrations/pkg-promptkit.md` to pin and link v0.5.0 and state
|
||||
the implemented dependency-level behavior: unset optional sampling controls
|
||||
are provider defaults. Do not document later stages as implemented.
|
||||
- Update any other canonical text that explicitly claims the dependency is
|
||||
v0.3.0, but defer descriptions of unimplemented v0.5 APIs.
|
||||
|
||||
- The pipeline retains one input, chunk plan, output encoder, run identity,
|
||||
checkpoint identity, failure boundary, worker budget, and provider scheduler.
|
||||
- Every selected module and validator is constructed before source parsing.
|
||||
- Steps schedule fixed extract, merge, and normalize lanes; this work must not
|
||||
introduce module-to-module calls or a general DAG scheduler.
|
||||
- Generated artifacts remain cloned operation-time context, never source
|
||||
evidence. Never emit their content, source material, credentials, or local
|
||||
paths in decisions, manifests, logs, or summaries.
|
||||
- Public ordering remains step order, lane ID, and source/chunk order as
|
||||
applicable. Refactoring must not expose completion order.
|
||||
- Tests must follow `docs/policy/testing.md`: protect observable recovery,
|
||||
orchestration, and CLI contracts without asserting private helper calls,
|
||||
exact prose, goroutine choreography, or full-document snapshots.
|
||||
- Update current-behavior documentation in the same stage that changes the
|
||||
corresponding behavior. Keep detailed implementation sequencing only here.
|
||||
### Tests And Validation
|
||||
|
||||
## Target Checkpoint Semantics
|
||||
- `go test ./internal/framework/llm ./internal/cli`
|
||||
- `go test ./...`
|
||||
- `go vet ./...`
|
||||
- `go build ./cmd/notarius`
|
||||
- `rg -n 'promptkit v0\.3\.0|promptkit@v0\.3\.0|PromptKit v0\.3\.0' .`
|
||||
- `git diff --check`
|
||||
|
||||
Use these definitions consistently in all stages:
|
||||
### Completion Criteria
|
||||
|
||||
- A **stage checkpoint** is internal resumable state for extract, merge, or
|
||||
normalize. Ordinary resume may continue to reuse this state progressively.
|
||||
- An **accepted lane artifact** is the one successful normalized artifact for a
|
||||
`(step ID, lane ID, normalizer module)` under the current checkpoint identity.
|
||||
It is the dependency exposed to a later step.
|
||||
- An unselected required predecessor satisfies selective recomputation when its
|
||||
accepted lane artifact can be validated and hydrated. Its extract and merge
|
||||
stage checkpoints are not prerequisites for that handoff.
|
||||
- A selected lane and its transitive dependents execute. They must never use
|
||||
accepted-artifact hydration to bypass forced execution.
|
||||
- If a required predecessor's accepted lane artifact is missing, rejected,
|
||||
corrupt, non-canonical, or incompatible, fail the run before any dependent
|
||||
lane starts. Do not implicitly rerun that predecessor.
|
||||
- Ordinary resume without `--recompute-step` keeps its existing progressive
|
||||
stage-reuse and cold-miss behavior.
|
||||
- The repository directly pins v0.5.0 and all default offline checks pass.
|
||||
- The profile-source fingerprint identifies the new upstream catalog without a
|
||||
brittle literal-hash test.
|
||||
- Current documentation no longer identifies v0.3.0 as the supported version.
|
||||
|
||||
## Stage 1: Decompose Runner And Checkpoint Orchestration
|
||||
## Stage 2: Execute One Frozen Prepared Snapshot
|
||||
|
||||
### Objective
|
||||
### Goal
|
||||
|
||||
Create explicit, testable ownership seams for ordered-step coordination,
|
||||
per-lane execution, and per-stage checkpoint handling without changing
|
||||
observable behavior.
|
||||
Make Notarius debug details and generation use one exact PromptKit preparation.
|
||||
|
||||
### Changes
|
||||
### Work
|
||||
|
||||
1. Keep `Runner.Run` as the pipeline-wide coordinator. Extract a small
|
||||
step-coordination helper responsible only for iterating prepared steps,
|
||||
building generated reference sets at each barrier, invoking the lane engine,
|
||||
and merging deterministic outcomes.
|
||||
2. Split `runLanes` in `internal/framework/pipeline/runner_concurrent.go` into
|
||||
helpers with these responsibilities:
|
||||
- initialize lane state and resolve extract checkpoint state in lane order;
|
||||
- run the bounded extract worker/continuation engine;
|
||||
- collect terminal lane results and choose failures deterministically; and
|
||||
- merge lane-local output into step output.
|
||||
Preserve the existing bounded channels, cancellation, drain behavior,
|
||||
chunk-first dispatch, lane ordering, and step barrier.
|
||||
3. Split `continueTypedLane` in
|
||||
`internal/framework/pipeline/runner_typed.go` into stage-specific merge and
|
||||
normalize helpers. Each helper should own dependency construction, checkpoint
|
||||
loading and canonical validation, execution/retry/validation when needed,
|
||||
checkpoint recording, debug envelopes, and its typed result. Use small
|
||||
result structs rather than long parallel return lists.
|
||||
4. Centralize repeated checkpoint-decision flow in one pipeline helper that can
|
||||
apply forced-execution policy, validate canonical stored artifacts, record
|
||||
the final observable decision, and return a contextual error. Do not yet
|
||||
change categories, reason codes, or required-predecessor semantics; Stages 2
|
||||
and 3 will change those deliberately.
|
||||
5. Keep stage-specific code where the data shapes genuinely differ. Do not
|
||||
introduce reflection, a generic stage state machine, or callbacks that hide
|
||||
the fixed extract/merge/normalize lifecycle.
|
||||
- Refactor `PromptKitClient.CompleteStructured` to call
|
||||
`PrepareExecution`, immediately defer `Discard`, obtain a caller-owned
|
||||
`Details` value, and execute with `RunPrepared`.
|
||||
- Preserve the existing Notarius request mapping, cancellation precedence,
|
||||
validation classification, raw structured bytes, response decoding,
|
||||
profile recording, usage reporting, and credential redaction.
|
||||
- Ensure every preparation, execution, validation, empty-result, and decode
|
||||
error retains useful Notarius prompt context without exposing prepared handle
|
||||
state or secrets.
|
||||
- Use `errors.As` to obtain `*promptkit.CapacityError` on admission rejection.
|
||||
Preserve `contracts.ErrLLMCapacityExceeded` as the stable classification and
|
||||
add a nonblank backend ID only to safe application-owned diagnostic context.
|
||||
Do not expose `promptkit.CapacityError` outside the LLM adapter.
|
||||
- Update `docs/internal/llm.md` and the implemented-mechanics portion of
|
||||
`docs/integrations/pkg-promptkit.md` to describe the single frozen execution
|
||||
snapshot and structured capacity adaptation.
|
||||
|
||||
### Tests
|
||||
### Tests And Validation
|
||||
|
||||
- Existing runner, checkpoint, barrier, ordering, cancellation, retry, debug,
|
||||
and D&D integration tests must pass without expectation changes except moves
|
||||
required by renamed private test fixtures.
|
||||
- Add no tests for helper boundaries or collaborator call counts. Add a narrow
|
||||
regression assertion only if the refactor exposes an observable behavior not
|
||||
already protected.
|
||||
- Adapt existing PromptKit client tests to the prepared-execution path.
|
||||
- Retain or add one behavioral test proving that the debug prompt details match
|
||||
the request actually passed to generation when a backing prompt source could
|
||||
otherwise change between independent preparations. Test the resulting
|
||||
snapshot consistency, not a private helper call count.
|
||||
- Retain capacity tests proving `errors.Is` reaches
|
||||
`contracts.ErrLLMCapacityExceeded`, the selected backend can appear in safe
|
||||
diagnostic context, and provider calls are not made after rejected
|
||||
admission.
|
||||
- Run `go test ./internal/framework/llm` and
|
||||
`go test -race ./internal/framework/llm`.
|
||||
- Run `go test ./...` and `git diff --check`.
|
||||
|
||||
### Completion Gate
|
||||
### Completion Criteria
|
||||
|
||||
Run `go test ./internal/framework/pipeline ./internal/framework/checkpoint
|
||||
./internal/modules/integration` and the pipeline race tests. Review the diff to
|
||||
confirm this stage changes structure only: serialized output, checkpoint state,
|
||||
decision values, failure selection, and module invocation behavior must remain
|
||||
unchanged.
|
||||
- `CompleteStructured` no longer calls independent `Prepare` and `Run`
|
||||
operations for one request.
|
||||
- Debug prompt material and generation result originate from the same frozen
|
||||
PromptKit snapshot.
|
||||
- Capacity remains a provider-neutral Notarius error classification.
|
||||
|
||||
## Stage 2: Make Checkpoint Decisions Typed And Observable
|
||||
## Stage 3: Replace Synthetic Profile Validation With Inspection
|
||||
|
||||
### Objective
|
||||
### Goal
|
||||
|
||||
Assign stable decision categories and reason codes explicitly, and retain the
|
||||
decision that causes a required-dependency failure.
|
||||
Validate profiles through PromptKit's exact profile-inspection boundary and
|
||||
centralize engine profile-source construction.
|
||||
|
||||
### Changes
|
||||
### Work
|
||||
|
||||
1. In `internal/framework/pipeline/checkpoint.go`, introduce string-backed
|
||||
internal types and constants for checkpoint decision categories and reason
|
||||
codes. Keep the current JSON strings and public artifact fields compatible.
|
||||
Categories remain exactly `executed`, `reused`, `forced_recompute`, and
|
||||
`dependency_invalidated`.
|
||||
2. Replace prose inspection in `internal/framework/checkpoint/loader.go` with an
|
||||
explicit decision constructor accepting category, reason code, and optional
|
||||
detail. Remove every `strings.Contains` classification branch. Assign a code
|
||||
at the validation site using this bounded vocabulary:
|
||||
- `loading_disabled`, `checkpoint_missing`, `checkpoint_path_invalid`,
|
||||
`checkpoint_read_failed`, and `checkpoint_decode_failed`;
|
||||
- `workspace_schema_incompatible` and `identity_mismatch`;
|
||||
- `stage_mismatch`, `step_mismatch`, `lane_mismatch`, `module_mismatch`, and
|
||||
`status_not_reusable`;
|
||||
- `dependency_mismatch` for the `dependency_invalidated` category;
|
||||
- `artifact_payload_invalid`, `artifact_digest_mismatch`,
|
||||
`artifact_codec_incompatible`, and `artifact_not_canonical`; and
|
||||
- `checkpoint_reused` and `accepted_artifact_reused` for successful reuse,
|
||||
and `recompute_step` for forced execution.
|
||||
Retain an existing code not listed here only when a current external or
|
||||
documented contract already depends on it.
|
||||
3. Keep detail human-readable, UTF-8, bounded, and sanitized through one helper.
|
||||
Detail may identify stage, step, lane, module, expected status, or schema
|
||||
version, but must not include artifact/reference content, source content,
|
||||
credentials, environment values, or local paths. Tests must assert codes and
|
||||
safety properties, not exact detail prose.
|
||||
4. Change the centralized runner decision flow from Stage 1 so the final loader
|
||||
or canonical-validation decision is recorded before returning a
|
||||
required-predecessor error. The contextual error must identify the step and
|
||||
lane and include the stable reason code; it must not interpolate unsafe
|
||||
loader detail.
|
||||
5. Propagate typed values without lossy conversion through checkpoint events,
|
||||
run manifests, debug summaries, and CLI diagnostics. Convert to strings only
|
||||
at existing serialized boundaries unless changing an internal field type is
|
||||
simpler and wire-compatible.
|
||||
6. Update the canonical current-behavior owners in the same change:
|
||||
`docs/internal/state.md` owns the internal decision flow, while
|
||||
`docs/operations.md` owns operator diagnosis and any operator-visible reason
|
||||
code table. Link rather than duplicate the table elsewhere.
|
||||
- Introduce a small provider-adapter-owned profile inspection or validation
|
||||
function in `internal/framework/llm`. Its public internal signature must use
|
||||
Notarius-owned configuration and result/error types rather than returning
|
||||
PromptKit types to the CLI.
|
||||
- Share the code that applies `profile_dir`, `profile_file`, and registered
|
||||
backend options between the production PromptKit engine and the inspection
|
||||
engine. Preserve the mutual-exclusion and local-backend rules.
|
||||
- Change CLI explicit-profile preflight to use `Engine.InspectProfile` through
|
||||
that LLM boundary.
|
||||
- Remove `profileCheckPromptID`, `profileCheckPromptFS`, the `testing/fstest`
|
||||
production dependency, and the synthetic `Prepare` request.
|
||||
- Preserve distinct, useful errors for an absent profile, invalid profile,
|
||||
unknown backend registration, cancellation, and invalid profile source.
|
||||
- Do not require `api_key_env` to be populated during configuration validation.
|
||||
Inspection may report credential requirements internally, but actual
|
||||
preparation remains responsible for credential availability before a model
|
||||
call.
|
||||
- Update current-behavior sections in `docs/internal/cli.md` and
|
||||
`docs/internal/llm.md`. Keep field definitions in `docs/config.md`.
|
||||
|
||||
### Tests
|
||||
### Tests And Validation
|
||||
|
||||
- Add a table in `internal/framework/checkpoint` that exercises one
|
||||
representative input per reason-code family and asserts category, code,
|
||||
bounded valid UTF-8 detail, and absence of supplied secret/path sentinels.
|
||||
- Add pipeline behavior tests proving a required-predecessor failure records the
|
||||
underlying missing, corrupt, incompatible, or dependency-invalidated
|
||||
decision before returning.
|
||||
- Retain focused manifest/debug serialization assertions for category and code;
|
||||
do not add exact-detail or full-manifest snapshots.
|
||||
- Replace synthetic-prompt tests with profile inspection tests covering:
|
||||
configured local backend success; missing local backend failure; absent
|
||||
profile; malformed profile; and an otherwise valid profile whose credential
|
||||
environment variable is intentionally unset.
|
||||
- Prove validation performs no provider HTTP call and remains offline.
|
||||
- Run `go test ./internal/framework/llm ./internal/cli` and `go test ./...`.
|
||||
- Run `git diff --check`.
|
||||
|
||||
### Completion Gate
|
||||
### Completion Criteria
|
||||
|
||||
Run `go test ./internal/framework/checkpoint ./internal/framework/pipeline
|
||||
./internal/core/artifacts ./internal/core/debugbundle ./internal/cli`. Search the
|
||||
production checkpoint package to confirm no decision category or reason code is
|
||||
derived from diagnostic prose.
|
||||
- No production synthetic profile-check prompt remains.
|
||||
- Profile validation uses the same ordinary profile source and backend
|
||||
registrations as execution.
|
||||
- Configuration validation succeeds for structurally valid profiles without
|
||||
reading credential values.
|
||||
|
||||
## Stage 3: Hydrate Required Predecessors From Accepted Normalized Artifacts
|
||||
## Stage 4: Add Application Fallback Profile Asset Plumbing
|
||||
|
||||
### Objective
|
||||
### Goal
|
||||
|
||||
Make selective recomputation enforce the lane-level accepted-artifact contract:
|
||||
a valid normalized producer artifact is sufficient even when its extract or
|
||||
merge stage cache is unavailable.
|
||||
Allow module families to register application-owned fallback profile YAML
|
||||
without placing domain policy in generic LLM code.
|
||||
|
||||
### Changes
|
||||
### Work
|
||||
|
||||
1. Extend the checkpoint loader contract with a dedicated accepted-normalized-
|
||||
artifact lookup. Implement it in the filesystem loader and every no-op or
|
||||
test implementation. The lookup receives step ID, lane ID, and normalizer
|
||||
module key and returns the normalized artifact, its warnings, and an explicit
|
||||
checkpoint decision. It must not require caller-supplied extract or merge
|
||||
dependency fingerprints.
|
||||
2. The filesystem lookup may reuse the existing normalize manifest and payload;
|
||||
do not add a second persistent copy. It is reusable only when all of the
|
||||
following hold:
|
||||
- workspace schema is v3 and the non-empty stored checkpoint identity matches
|
||||
the current invocation identity;
|
||||
- stage, step, lane, normalizer module, and successful status match;
|
||||
- the manifest output digest matches the payload; and
|
||||
- the payload can subsequently be validated through the registered artifact
|
||||
codec.
|
||||
Skipping caller-supplied merge dependencies is safe only because the matched
|
||||
non-empty checkpoint identity already binds the current input, resolved
|
||||
topology and configuration, external references, runtime overrides, LLM
|
||||
profiles, and component semantic fingerprints. Do not weaken or omit that
|
||||
identity check.
|
||||
A loader lacking a verifiable current identity must return an unavailable
|
||||
decision rather than perform accepted-artifact reuse.
|
||||
3. Add a pipeline hydration helper that decodes the returned serialized
|
||||
artifact through the producer's registered codec, re-encodes it canonically,
|
||||
and requires exact artifact kind, schema ID/name/version/digest, media type,
|
||||
content bytes, and content digest. Return a runner-owned clone with producer
|
||||
step, lane, module, and source identity. Any mismatch is an explicit bounded
|
||||
decision and the bytes never reach a consumer.
|
||||
4. Before normal execution of a lane marked in `RequireReusableLanes`, use the
|
||||
accepted-artifact lookup:
|
||||
- on success, mark the lane terminal without invoking extract, merge,
|
||||
normalize, or their validators;
|
||||
- append the accepted normalized output in the normal deterministic location
|
||||
and restore only the warnings stored with that normalized checkpoint;
|
||||
- record one `reused` normalize decision with a stable accepted-output reuse
|
||||
reason code; do not synthesize extract/merge decisions, warnings, or
|
||||
rejections that were not loaded; and
|
||||
- allow the ordinary step barrier and generated handoff code to consume that
|
||||
output exactly as it consumes a freshly executed output.
|
||||
5. On missing, rejected, corrupt, non-canonical, or incompatible accepted state,
|
||||
record the decision and fail the run before the dependent step begins. Do
|
||||
not fall back to stage execution. Forced lanes must bypass this hydration
|
||||
path and execute normally.
|
||||
6. Leave ordinary resume unchanged when no lane is marked
|
||||
`RequireReusableLanes`: it may reuse or recompute extract, merge, and
|
||||
normalize progressively under the existing cold-miss rules.
|
||||
7. Keep generated-reference fingerprints and provenance unchanged. Hydrating a
|
||||
byte-identical producer must yield the same canonical handoff digest and
|
||||
downstream dependency fingerprint as fresh execution.
|
||||
8. Update `docs/internal/pipeline.md`, `docs/internal/state.md`, and
|
||||
`docs/operations.md` in this stage to distinguish progressive stage reuse
|
||||
from accepted normalized-artifact hydration and to document the fail-rather-
|
||||
than-rerun rule for invalid required predecessors.
|
||||
- Extend `internal/framework/llm.AssetRegistry` with a separate fallback
|
||||
profile source collection, registration method, flattened filesystem, and
|
||||
safe content digest.
|
||||
- Reuse the existing asset-source path validation and flattening behavior where
|
||||
appropriate. Reject invalid roots, unreadable assets, and duplicate flattened
|
||||
paths. Do not parse PromptKit profile YAML in Notarius.
|
||||
- Add `promptkit.WithFallbackProfileFS` to production engine options only when
|
||||
at least one fallback profile source is registered.
|
||||
- Supply the identical assembled fallback source to the profile-inspection
|
||||
engine. Adjust CLI composition so pipeline-aware profile validation can use
|
||||
the production LLM asset registry without exposing PromptKit types.
|
||||
- Extend profile-source checkpoint identity to include the exact fallback
|
||||
profile asset digest in addition to the PromptKit catalog marker and operator
|
||||
source. Keep the resulting fingerprint hash-only and path/content/credential
|
||||
free.
|
||||
- Keep operator source precedence owned by PromptKit. Do not implement profile
|
||||
merging or duplicate PromptKit source resolution in Notarius.
|
||||
- Update `docs/internal/llm.md` only for the new implemented generic asset and
|
||||
fingerprint mechanics. No domain fallback exists until Stage 5.
|
||||
|
||||
### Tests
|
||||
### Tests And Validation
|
||||
|
||||
- At the pipeline/checkpoint boundary, create a valid producer normalize
|
||||
checkpoint while omitting or corrupting its extract and merge checkpoints.
|
||||
Select a later step for recomputation and prove the producer hydrates, no
|
||||
producer operation or validator runs, and the dependent receives the exact
|
||||
canonical artifact.
|
||||
- Cover missing, rejected-status, corrupt, non-canonical, wrong-codec-identity,
|
||||
and wrong-content-digest normalize state. Assert failure and the recorded
|
||||
stable decision before any consumer invocation.
|
||||
- Prove forced producers execute instead of hydrating, while a reusable
|
||||
predecessor and an unrelated lane remain reusable.
|
||||
- Prove fresh and hydrated producer outputs create identical generated
|
||||
provenance and downstream checkpoint fingerprints without exposing content
|
||||
or paths.
|
||||
- Add focused AssetRegistry tests for successful flattening, invalid roots,
|
||||
duplicate paths, and hash changes when fallback bytes change.
|
||||
- Add adapter-level tests showing that the fallback filesystem reaches both
|
||||
execution construction and inspection construction.
|
||||
- Extend checkpoint tests to prove fallback content changes profile-source
|
||||
identity without exposing raw YAML or paths. Use relational comparisons, not
|
||||
a fixed hash literal.
|
||||
- Run `go test ./internal/framework/llm ./internal/cli` and `go test ./...`.
|
||||
- Run `git diff --check`.
|
||||
|
||||
### Completion Gate
|
||||
### Completion Criteria
|
||||
|
||||
Run `go test ./internal/framework/checkpoint ./internal/framework/pipeline
|
||||
./internal/modules/integration` and the corresponding race tests. Manually
|
||||
inspect one failed test fixture to confirm the accepted artifact remains on
|
||||
disk but its bytes do not appear in the manifest, decision detail, debug
|
||||
summary, or error.
|
||||
- Generic plumbing can carry application fallback profiles while remaining
|
||||
unaware of D&D IDs or model settings.
|
||||
- Inspection, execution, and checkpoint identity use the same fallback asset
|
||||
source.
|
||||
|
||||
## Stage 4: Complete Recompute CLI And Recovery Acceptance Coverage
|
||||
## Stage 5: Adopt The D&D `dnd-extraction` Fallback
|
||||
|
||||
### Objective
|
||||
### Goal
|
||||
|
||||
Exercise the operator-visible recomputation contract through stable boundaries
|
||||
and close the acceptance-test omissions from the original plan.
|
||||
Give the D&D module family one stable embedded workload profile that operators
|
||||
can replace.
|
||||
|
||||
### Changes
|
||||
### Work
|
||||
|
||||
1. Add one table-driven CLI contract test for `--recompute-step` covering:
|
||||
- a valid explicit step and the implicit `default` step;
|
||||
- repeated flags and an empty or unknown step ID;
|
||||
- use without `--resume`;
|
||||
- use when checkpoint recording is disabled; and
|
||||
- combination with `--only`.
|
||||
Assert exit classification and stable identifying fragments or reason codes,
|
||||
not complete prose.
|
||||
2. Add one filesystem-backed CLI execution test using deterministic fake
|
||||
modules and a three-step generated dependency chain plus one unrelated lane:
|
||||
- perform a fresh checkpointed run;
|
||||
- remove or corrupt only the required producer's extract and merge state
|
||||
inside `t.TempDir()`, leaving its normalize artifact valid;
|
||||
- resume with the middle step selected;
|
||||
- assert the selected lane and transitive dependents execute, the predecessor
|
||||
hydrates without module calls, the unrelated lane reuses, and output and
|
||||
decision ordering are deterministic; and
|
||||
- invalidate the producer normalize state in a subcase and assert the command
|
||||
fails before dependent execution with the bounded decision preserved.
|
||||
3. Add or extend one generic two-step pipeline test proving handoff succeeds
|
||||
both from fresh producer execution and from accepted normalized-artifact
|
||||
hydration. Keep this at the pipeline boundary if the CLI execution test
|
||||
already proves flag wiring; do not duplicate every CLI case end to end.
|
||||
4. Review existing recomputation policy tests. Retain the pure closure test
|
||||
because it protects transitive selection, but remove or consolidate any new
|
||||
test that merely repeats the CLI or pipeline behavior above.
|
||||
5. Make only production changes revealed as necessary by these contract tests;
|
||||
do not add new flag semantics or broaden the feature roadmap.
|
||||
- Add a D&D-owned embedded PromptKit profile asset with ID `dnd-extraction`
|
||||
under `internal/modules/dnd`. Use the exact baseline defined in
|
||||
`promptkit.md`: OpenRouter, `openai/gpt-5.6-luna`, no explicit reasoning
|
||||
effort, a 240-second timeout, flex service tier, and no selected temperature,
|
||||
token limit, or `top_p`. The omitted reasoning value intentionally allows
|
||||
OpenAI's backend to apply its `medium` default.
|
||||
- Register the profile filesystem from the D&D registrar through the generic
|
||||
fallback profile asset boundary. Keep D&D policy out of
|
||||
`internal/framework/llm` and the CLI composition root.
|
||||
- Change every maintained D&D LLM prompt definition—including scene chunking,
|
||||
all D&D extractors, and NPC normalization—from the model-named default to
|
||||
`default_profile: dnd-extraction`.
|
||||
- Add an integration-level profile-resolution test proving that:
|
||||
- the fallback resolves when no operator source defines the ID;
|
||||
- a valid operator profile with the same ID wins completely; and
|
||||
- an invalid matching operator profile fails rather than falling through.
|
||||
- Test through Notarius's assembled production assets and PromptKit boundary;
|
||||
do not duplicate every upstream source-precedence case.
|
||||
- Update the implemented profile ownership and prompt-default behavior in
|
||||
`docs/internal/dnd.md`, `docs/internal/llm.md`, and
|
||||
`docs/integrations/pkg-promptkit.md`. Defer the complete operator walkthrough
|
||||
and examples to Stage 10.
|
||||
|
||||
### Completion Gate
|
||||
### Tests And Validation
|
||||
|
||||
Run `go test ./internal/cli ./internal/framework/pipeline
|
||||
./internal/framework/checkpoint`, then run the CLI, pipeline, and checkpoint
|
||||
packages under the race detector. Confirm no test asserts internal loader call
|
||||
counts, exact decision detail, filesystem layout outside a temp workspace, or
|
||||
the exact length of any prompt or prefix.
|
||||
- Run focused D&D prompt preparation tests and the production composition
|
||||
tests.
|
||||
- Run `go test ./internal/modules/dnd/... ./internal/framework/llm
|
||||
./internal/cli`.
|
||||
- Run `go test ./...`.
|
||||
- Verify `rg -n 'default_profile: gemini-2-flash' internal/modules/dnd`
|
||||
returns no matches.
|
||||
- Run `git diff --check`.
|
||||
|
||||
## Stage 5: Current Documentation And Release Verification
|
||||
### Completion Criteria
|
||||
|
||||
### Objective
|
||||
- All maintained D&D prompts use the application-owned logical profile ID.
|
||||
- The fallback works without an operator profile and remains authoritatively
|
||||
overridable by a matching valid operator definition.
|
||||
|
||||
Reconcile all canonical documentation after the staged changes and verify the
|
||||
complete feature against policy and roadmap.
|
||||
## Stage 6: Introduce Module Execution-Class Metadata
|
||||
|
||||
### Changes
|
||||
### Goal
|
||||
|
||||
1. Re-read `docs/internal/pipeline.md`, `docs/internal/state.md`,
|
||||
`docs/operations.md`, and `docs/cli.md` against the final code. Correct any
|
||||
stale statements left by Stages 1 through 4 without duplicating their
|
||||
canonical contracts.
|
||||
2. Ensure internal documentation describes the coordinator, lane engine, stage
|
||||
checkpoint flow, and accepted-artifact validation at the responsibility
|
||||
level without listing volatile private helper names.
|
||||
3. Confirm the operator-visible reason-code table has one canonical owner and
|
||||
other documents link to it rather than maintaining parallel copies.
|
||||
4. Remove stale implementation claims exposed by this work and ensure the
|
||||
feature roadmap continues to describe target policy rather than task
|
||||
sequencing.
|
||||
Make each production module's ability to use an LLM statically discoverable
|
||||
without yet changing profile inheritance.
|
||||
|
||||
### Verification
|
||||
### Work
|
||||
|
||||
Run the repository-prescribed commands from `docs/development.md`:
|
||||
- Add `ExecutionClass contracts.ExecutionClass` to `pipeline.ModuleSpec` and
|
||||
preserve it through normalization, cloning, catalogs, registries, JSON/debug
|
||||
views, and lookup helpers.
|
||||
- In this transitional stage only, allow an omitted execution class to
|
||||
normalize to deterministic so existing test-only fixtures can be migrated in
|
||||
Stage 7 without breaking the repository midway.
|
||||
- Explicitly classify every production module:
|
||||
- D&D scene chunking, every D&D extractor, and D&D NPC normalization as
|
||||
`llm_backed`;
|
||||
- all other current production input, chunk, merge, normalize, and output
|
||||
modules as `deterministic`.
|
||||
- Update production module specification tests and production catalog tests to
|
||||
assert the semantic class alongside stage, artifact kind, and capabilities.
|
||||
- Add catalog lookup support needed by later resolution to retrieve a selected
|
||||
module's execution class by stage and key without constructing it.
|
||||
- Do not implement pipeline-level profile inheritance or reject deterministic
|
||||
profiles yet.
|
||||
- Update `docs/internal/modules.md` and `docs/internal/dnd.md` to identify
|
||||
execution class as registered module metadata, while noting only implemented
|
||||
uses.
|
||||
|
||||
### Tests And Validation
|
||||
|
||||
- Run module registration/spec tests across generic, Seriatim, and D&D
|
||||
families.
|
||||
- Run `go test ./internal/framework/pipeline ./internal/modules/...`.
|
||||
- Run `go test ./...` and `git diff --check`.
|
||||
|
||||
### Completion Criteria
|
||||
|
||||
- Every production module has an explicit correct execution class.
|
||||
- Catalog consumers can retrieve that class without a concrete module
|
||||
instance.
|
||||
- Test-only omitted classes remain the only temporary compatibility behavior.
|
||||
|
||||
## Stage 7: Enforce Execution Metadata And Remove Runtime Probing
|
||||
|
||||
### Goal
|
||||
|
||||
Finish the execution-class contract so missing metadata cannot cause future
|
||||
profile drift.
|
||||
|
||||
### Work
|
||||
|
||||
- Update every framework, CLI, and integration test module specification to
|
||||
declare an explicit execution class appropriate to the fake behavior.
|
||||
- Change module-spec validation so an empty or unsupported execution class is a
|
||||
registration error. Remove the transitional deterministic default from
|
||||
Stage 6.
|
||||
- Replace the chunk runner's special `ChunkExecutionClassProvider` probe with
|
||||
specification-derived behavior. Remove the now-redundant provider interface,
|
||||
implementation methods, and tests when they have no remaining consumer.
|
||||
- Ensure chunk producer provenance remains unchanged: it records a non-empty
|
||||
effective binding profile for an LLM-backed chunker, while a deterministic
|
||||
chunker records no profile. A profile selected only through the prompt
|
||||
default remains represented by PromptKit's actual-profile manifest rather
|
||||
than being invented as an explicit chunk binding.
|
||||
- Review helper constructors and fixtures for opportunities to set execution
|
||||
class once without obscuring the class under test. Do not introduce an
|
||||
elaborate test-spec framework.
|
||||
- Update internal documentation if the removal changes any described runtime
|
||||
mechanics.
|
||||
|
||||
### Tests And Validation
|
||||
|
||||
- Add or retain focused registration tests for missing and invalid execution
|
||||
classes.
|
||||
- Retain chunk-plan provenance tests for LLM-backed and deterministic
|
||||
chunkers.
|
||||
- Run `go test ./internal/framework/pipeline ./internal/modules/...`.
|
||||
- Run `go test ./...`, `go vet ./...`, and `git diff --check`.
|
||||
|
||||
### Completion Criteria
|
||||
|
||||
- No registered module specification relies on an implicit execution class.
|
||||
- Pipeline metadata, not a concrete runtime type assertion, owns module
|
||||
execution classification.
|
||||
|
||||
## Stage 8: Resolve Programmatic Pipeline Profile Defaults
|
||||
|
||||
### Goal
|
||||
|
||||
Implement profile inheritance and precedence inside the pipeline resolver
|
||||
before exposing the field through YAML configuration.
|
||||
|
||||
### Work
|
||||
|
||||
- Add an optional trimmed `LLMProfile` field to
|
||||
`pipeline.PipelineProfile`. Add a non-empty runtime override field to
|
||||
`pipeline.ResolveOptions` so all precedence decisions occur in the resolver
|
||||
rather than through pre-resolution mutation.
|
||||
- After module selection, `--only` filtering, default validator-chain
|
||||
selection, and validator compatibility resolution, apply effective profiles
|
||||
to every selected input, chunk, extract, merge, normalize, output, and
|
||||
validator binding according to the precedence in `promptkit.md`.
|
||||
- Apply profiles only when the selected module or validator execution class is
|
||||
`llm_backed`.
|
||||
- Reject a binding-specific `llm_profile` on any deterministic module or
|
||||
validator. Do not reject or inspect an unused pipeline default when no
|
||||
selected LLM-backed binding consumes it.
|
||||
- Leave an LLM-backed binding empty when no CLI, binding, or pipeline profile is
|
||||
selected so PromptKit can use the prompt's `default_profile`.
|
||||
- Store the effective values on resolved bindings before digest construction.
|
||||
Do not add a second inheritance decision to execution.
|
||||
- Ensure semantically equivalent repeated binding profiles and one inherited
|
||||
default produce the same resolved pipeline digest. Ensure any changed
|
||||
effective profile changes the digest.
|
||||
- Do not modify file configuration or CLI parsing in this stage.
|
||||
|
||||
### Tests And Validation
|
||||
|
||||
- Add pipeline package tests for the complete precedence matrix:
|
||||
runtime override; binding-specific exception; pipeline default; prompt
|
||||
fallback; and deterministic bindings.
|
||||
- Cover default and explicitly configured validator chains, all relevant stage
|
||||
categories, `--only` lane selection, unused defaults, deterministic-profile
|
||||
rejection, and semantic digest equivalence.
|
||||
- Prefer table-driven package-level tests over assertions on private traversal
|
||||
helpers.
|
||||
- Run `go test ./internal/framework/pipeline` and `go test ./...`.
|
||||
- Run `git diff --check`.
|
||||
|
||||
### Completion Criteria
|
||||
|
||||
- Programmatic pipelines resolve one canonical effective profile policy.
|
||||
- Only LLM-backed resolved bindings can contain a profile.
|
||||
- Runtime override, binding, pipeline, and prompt precedence is unambiguous and
|
||||
digest-stable.
|
||||
|
||||
## Stage 9: Expose Pipeline Defaults Through Configuration And CLI
|
||||
|
||||
### Goal
|
||||
|
||||
Make the profile-default workflow available to operators while preserving
|
||||
validation and override behavior.
|
||||
|
||||
### Work
|
||||
|
||||
- Add optional `pipelines.<id>.llm_profile` support to the version 4 file
|
||||
configuration model. Use presence-aware decoding so an explicitly set blank
|
||||
value is rejected, while omission remains valid.
|
||||
- Preserve the field through file application, configuration cloning,
|
||||
effective configuration, and programmatic profile copies without aliasing or
|
||||
trimming drift.
|
||||
- Remove `applyLLMProfileOverride`. Pass the CLI override through the resolver's
|
||||
runtime-override input so deterministic bindings are never populated.
|
||||
- Update effective profile-ID collection to cover every selected LLM-backed
|
||||
module stage and LLM-backed validator, including future LLM-backed input and
|
||||
output modules. Do not inspect deterministic or unselected profiles.
|
||||
- Ensure `run`, `config validate --pipeline`, resume/checkpoint identity, and
|
||||
relevant dry preflight paths all use the same resolved effective profiles.
|
||||
- Preserve `--llm-profile` as the highest-precedence non-empty run-wide
|
||||
override and preserve binding-specific profiles as exceptions when no CLI
|
||||
override is present.
|
||||
- Do not increment the configuration version.
|
||||
- Update current configuration and CLI contracts in `docs/config.md` and
|
||||
`docs/cli.md` in the same stage. Link to operations for the deployment
|
||||
workflow rather than duplicating it prematurely.
|
||||
|
||||
### Tests And Validation
|
||||
|
||||
- Add file-config tests for omission, trimming, explicit blank rejection,
|
||||
unknown-key behavior, cloning, and round-trip application.
|
||||
- Add effective-config and CLI contract tests for precedence, LLM-only
|
||||
application, inherited-profile inspection failure before factory execution,
|
||||
`--only`, and digest changes.
|
||||
- Retain offline operation and do not require credentials for
|
||||
`config validate --pipeline`.
|
||||
- Run `go test ./internal/core/config ./internal/framework/pipeline
|
||||
./internal/cli`.
|
||||
- Run `go test ./...`, `go vet ./...`, and `git diff --check`.
|
||||
|
||||
### Completion Criteria
|
||||
|
||||
- Operators can select `dnd-extraction` once per pipeline.
|
||||
- Configuration and CLI paths share the resolver's precedence policy.
|
||||
- Unknown effective profiles fail preflight, while deterministic and unused
|
||||
profiles do not cause spurious inspection.
|
||||
|
||||
## Stage 10: Complete Operator Documentation, Examples, And Decision Record
|
||||
|
||||
### Goal
|
||||
|
||||
Make the implemented workflow understandable, copyable, and maintainable
|
||||
without duplicating canonical facts.
|
||||
|
||||
### Work
|
||||
|
||||
- Create an ADR using the next sequential number for the durable decision to
|
||||
use workload-oriented pipeline defaults with operator-overridable application
|
||||
fallback profiles. Record context, decision, alternatives, and consequences;
|
||||
do not turn the ADR into a field reference or implementation log.
|
||||
- Complete `docs/config.md` as the canonical owner of profile-source fields,
|
||||
`pipelines.<id>.llm_profile`, validation, and precedence.
|
||||
- Complete `docs/operations.md` with an operator workflow that distinguishes
|
||||
Notarius embedded prompts, Notarius fallback profiles, PromptKit built-ins,
|
||||
and deployment filesystem profiles. Include production/development/local use
|
||||
of the same `dnd-extraction` ID, credential handling, absolute-path guidance,
|
||||
and the fact that current relative profile paths use the process working
|
||||
directory rather than the configuration file's directory.
|
||||
- Complete `docs/integrations/pkg-promptkit.md` with the v0.5.0 boundary,
|
||||
prepared execution, inspection, fallback and ordinary source precedence,
|
||||
optional provider controls, capacity adaptation, and compatibility policy.
|
||||
- Update `docs/internal/configuration.md`, `docs/internal/pipeline.md`,
|
||||
`docs/internal/cli.md`, `docs/internal/llm.md`, `docs/internal/modules.md`, and
|
||||
`docs/internal/dnd.md` only for their owned implementation details. Link to
|
||||
canonical configuration, operations, and upstream format contracts rather
|
||||
than restating them.
|
||||
- Keep exactly the existing two D&D configuration examples. Add
|
||||
`llm_profile: dnd-extraction` to the minimal and complete pipelines and remove
|
||||
the now-redundant model-named binding override from the complete example.
|
||||
- Add one secret-free maintained operator profile at
|
||||
`examples/profiles/dnd-extraction.yml`. It should be a complete valid profile
|
||||
for the same logical ID and may mirror the embedded baseline; its purpose is
|
||||
to demonstrate file ownership and format, not claim automatic environment
|
||||
detection. Link it from the configuration and operations documentation.
|
||||
- If the complete example selects the external profile file, use a path that
|
||||
is valid for the documented repository-root invocation and explicitly note
|
||||
the working-directory rule. Keep the minimal example dependent only on the
|
||||
embedded fallback.
|
||||
- Add or extend maintained-example validation so both configuration examples
|
||||
and the profile YAML are checked without generation or credentials.
|
||||
- Remove the now-implemented `Pipeline-Level LLM Profile Defaults` section from
|
||||
`docs/roadmap/future.md`. Preserve the unrelated deterministic session and
|
||||
concurrency items.
|
||||
- Do not delete `promptkit.md` or this implementation plan during the feature
|
||||
implementation; retire them only after post-implementation review.
|
||||
|
||||
### Tests And Validation
|
||||
|
||||
- Run maintained example/configuration tests and relevant CLI help/parser
|
||||
tests.
|
||||
- Run `go test ./...`.
|
||||
- Run `rg -n 'gemini-2-flash' examples docs` and review every remaining match
|
||||
for intentional model-policy or historical context.
|
||||
- Run `rg -n 'v0\.3\.0|profileCheckPrompt|applyLLMProfileOverride' .` and resolve
|
||||
stale production or current-documentation matches.
|
||||
- Verify all new links and `git diff --check`.
|
||||
|
||||
### Completion Criteria
|
||||
|
||||
- Every current fact has one canonical documentation owner.
|
||||
- Operators can distinguish and deploy all profile layers without reading Go
|
||||
source.
|
||||
- Both maintained configurations and the maintained external profile are valid,
|
||||
secret-free, and tested offline.
|
||||
- Implemented profile work no longer remains in `future.md`.
|
||||
|
||||
## Stage 11: Final Verification And Quality Review
|
||||
|
||||
### Goal
|
||||
|
||||
Verify the complete migration as one integrated change and correct only defects
|
||||
or omissions found during that review.
|
||||
|
||||
### Work
|
||||
|
||||
- Review the final diff against every acceptance criterion in `promptkit.md`.
|
||||
- Confirm provider-specific PromptKit types remain inside the LLM integration
|
||||
boundary and D&D policy remains inside the D&D module family.
|
||||
- Confirm execution and inspection receive identical ordinary, fallback, and
|
||||
backend configuration.
|
||||
- Confirm no paths, profile YAML, endpoints, credentials, or prepared handle
|
||||
state leak into fingerprints or ordinary diagnostics.
|
||||
- Confirm all production module specs have explicit correct execution classes
|
||||
and every resolved deterministic binding is profile-free.
|
||||
- Confirm prompt default, pipeline default, binding override, and CLI override
|
||||
behavior through representative assembled configurations.
|
||||
- Review tests for redundancy and remove obsolete synthetic-prompt,
|
||||
runtime-probe, exact-hash, or duplicated upstream-behavior tests superseded by
|
||||
stronger contract tests.
|
||||
- Perform an optional manual D&D quality comparison if credentials and an
|
||||
evaluation transcript are deliberately supplied. Record no private input or
|
||||
credential material, and do not make this comparison a completion gate.
|
||||
|
||||
### Validation Commands
|
||||
|
||||
```sh
|
||||
gofmt -w <changed-go-files>
|
||||
go test ./...
|
||||
go test -race ./internal/framework/llm ./internal/core/config ./internal/framework/pipeline ./internal/cli
|
||||
go vet ./...
|
||||
go build ./cmd/notarius
|
||||
git diff --check
|
||||
```
|
||||
|
||||
Also run race tests for the pipeline, checkpoint, CLI state, D&D NPC registry,
|
||||
and D&D integration packages. Validate maintained examples/configurations
|
||||
through the production catalog. Review one fresh run, ordinary resumed run,
|
||||
forced-recompute run, hydrated-predecessor run, and invalid-predecessor failure
|
||||
for deterministic ordering, correct decisions, bounded provenance, and absence
|
||||
of generated content, secrets, or local paths in manifests and diagnostics.
|
||||
Also run focused stale-contract searches:
|
||||
|
||||
### Completion Gate
|
||||
```sh
|
||||
rg -n 'gitea.maximumdirect.net/eric/promptkit v0\.3\.0|PromptKit v0\.3\.0' .
|
||||
rg -n 'default_profile: gemini-2-flash|profileCheckPrompt|applyLLMProfileOverride' internal docs examples
|
||||
```
|
||||
|
||||
The follow-up is complete only when every finding summarized above is protected
|
||||
by a stable behavioral test, the current documentation matches the corrected
|
||||
implementation, the full verification suite passes, and the worktree contains
|
||||
no temporary adapters or TODOs introduced by these stages.
|
||||
Review any matches rather than deleting intentional historical references
|
||||
blindly.
|
||||
|
||||
### Completion Criteria
|
||||
|
||||
- All automated checks pass offline and without real credentials.
|
||||
- The implemented behavior matches `promptkit.md` with no known architecture,
|
||||
provenance, checkpoint, profile-precedence, or documentation gap.
|
||||
- Any optional live evaluation is clearly separate from correctness testing.
|
||||
|
||||
## Open Questions
|
||||
|
||||
None. The feature roadmap and this plan fix the required product and
|
||||
architecture choices. If implementation reveals that accepted normalized
|
||||
artifacts cannot be validated safely without a persistent format change, stop
|
||||
and amend this plan rather than weakening identity, codec, or content
|
||||
validation.
|
||||
None. The roadmap decisions are sufficient to implement every stage without an
|
||||
additional product or architecture choice.
|
||||
|
||||
@@ -1,339 +0,0 @@
|
||||
# Scope: Ordered Pipeline Steps
|
||||
|
||||
## Status
|
||||
|
||||
Implemented. This document preserves the bounded feature policy, architecture
|
||||
choices, acceptance criteria, and exclusions. Current behavior belongs in the
|
||||
canonical [CLI](../cli.md), [Configuration](../config.md),
|
||||
[Operations](../operations.md), and [internal pipeline](../internal/pipeline.md)
|
||||
documentation rather than in this roadmap.
|
||||
|
||||
## Policy Recommendation
|
||||
|
||||
Treat ordered pipeline steps, generated artifact references, and
|
||||
dependency-aware checkpoint reuse as one coherent platform capability. The D&D
|
||||
proving workflow produces accepted normalized NPC output first and then supplies
|
||||
it to spell extraction, combat-turn extraction, and combat-turn normalization.
|
||||
|
||||
## Intended Outcome
|
||||
|
||||
A configured pipeline may contain multiple ordered steps while retaining one
|
||||
pipeline-wide input, chunk plan, output, worker budget, LLM scheduler, run
|
||||
manifest, and failure boundary. Every artifact lane still follows the fixed
|
||||
extract, validate, merge, validate, normalize, and validate lifecycle. Steps
|
||||
add explicit barriers between groups of lanes; they do not create arbitrary
|
||||
stage graphs.
|
||||
|
||||
An accepted normalized artifact from an earlier step may be bound explicitly
|
||||
to declared reference slots in a later step. The framework remains
|
||||
domain-neutral, and generated references remain contextual material rather than
|
||||
source evidence.
|
||||
|
||||
## Fixed Product And Architecture Decisions
|
||||
|
||||
### Pipeline shape
|
||||
|
||||
- Input parsing and chunk planning remain pipeline-wide and execute once.
|
||||
- A step contains one or more artifact lanes. Step order is configuration order.
|
||||
- Lanes within a step remain independent and may use the existing bounded
|
||||
concurrency model.
|
||||
- A later step cannot begin until every lane in the current step is terminal
|
||||
and every generated artifact it requires is accepted and available.
|
||||
- Public artifact and failure ordering is step order followed by deterministic
|
||||
lane and source-chunk order, never completion order.
|
||||
- Output encoding occurs once, after every step succeeds.
|
||||
- This is not an arbitrary DAG, a general workflow language, concurrent
|
||||
cross-lane reconciliation, or permission for modules to invoke other modules.
|
||||
|
||||
### Configuration model
|
||||
|
||||
Existing single-step pipelines remain valid. A top-level `artifacts` map is
|
||||
treated as an implicit step with stable ID `default`. A pipeline may configure
|
||||
either `artifacts` or `steps`, but not both. Explicit steps must be non-empty
|
||||
and have unique, trimmed, non-empty IDs. Artifact lane IDs must remain unique
|
||||
across the entire pipeline so output paths, selectors, manifests, errors, and
|
||||
checkpoint scopes remain unambiguous.
|
||||
|
||||
The target configuration shape is:
|
||||
|
||||
```yaml
|
||||
pipelines:
|
||||
dnd-session:
|
||||
input: seriatim
|
||||
chunk: generic
|
||||
steps:
|
||||
- id: identify-npcs
|
||||
artifacts:
|
||||
npcs:
|
||||
extract: dnd/npcs
|
||||
normalize: dnd/npcs
|
||||
- id: grounded-events
|
||||
references:
|
||||
npcs:
|
||||
artifact:
|
||||
step: identify-npcs
|
||||
lane: npcs
|
||||
artifacts:
|
||||
spells:
|
||||
extract: dnd/spells
|
||||
normalize: dnd/spells
|
||||
combat:
|
||||
extract: dnd/combat-turns
|
||||
normalize: dnd/combat-turns
|
||||
output: json
|
||||
```
|
||||
|
||||
Existing scalar reference values continue to represent external file paths.
|
||||
The structured `artifact` form identifies accepted normalized output from one
|
||||
earlier step and lane. Generated artifact bindings are allowed at step scope or
|
||||
at an individual module target; they are not inferred from module keys, lane
|
||||
names, slot names, or domain knowledge.
|
||||
|
||||
A step-scoped reference applies automatically to every selected target in that
|
||||
step that declares the slot. In the example, one `npcs` binding reaches the
|
||||
spell extractor plus the combat extractor and normalizer. A target-local
|
||||
binding is used when only one module should consume the artifact.
|
||||
|
||||
Pipeline-level external references remain defaults. Step-local external
|
||||
references override pipeline-level external defaults, and target-local
|
||||
external references retain their existing precedence. A generated reference
|
||||
and an external reference may not resolve to the same effective target slot;
|
||||
configuration or a runtime override that creates that conflict is invalid.
|
||||
Likewise, a step-scoped and target-local generated binding cannot both target
|
||||
the same effective slot.
|
||||
|
||||
Each effective target slot accepts at most one producer. One producer may fan
|
||||
out to multiple compatible slots in a later step. Aggregating several producer
|
||||
artifacts into one slot is outside this scope.
|
||||
|
||||
Reference-slot specs gain optional generated-artifact compatibility metadata.
|
||||
A generated binding is allowed only when the consumer slot declares the
|
||||
producer's artifact kind; the producer's registered codec supplies the exact
|
||||
schema identity and media type used for the handoff. The D&D `npcs` consumer
|
||||
slots declare the normalized NPC-list artifact kind. Existing external-file
|
||||
slots and bindings retain their current behavior and do not acquire an artifact
|
||||
kind merely because their bytes happen to decode as one.
|
||||
|
||||
### Resolution and preparation
|
||||
|
||||
Resolution validates the complete ordered structure before source processing.
|
||||
It must reject duplicate identities, missing producers, same-step or forward
|
||||
references, undeclared slots, reference conflicts, and incompatible artifact
|
||||
kind, schema, media type, or cardinality constraints that are statically
|
||||
discoverable. Size is checked when canonical producer bytes exist at handoff.
|
||||
Ordered steps make cycles structurally impossible; resolution must not
|
||||
introduce a general graph scheduler to rediscover their order.
|
||||
|
||||
The resolved pipeline and its digest include step order, step IDs, lane
|
||||
membership, generated-reference topology, producer identity, consumer targets,
|
||||
and existing module and validator policy. Cloning, redaction, canonical JSON,
|
||||
debug summaries, and manifests preserve the same structure without reference
|
||||
content or secrets.
|
||||
|
||||
All modules and validators are still selected, option-validated, and
|
||||
constructed before source parsing. Generated content cannot be supplied during
|
||||
construction because it does not exist yet. The framework therefore augments
|
||||
the existing operation-request `References` at the step boundary. Consumers
|
||||
that currently assume an NPC registry is construction-only must accept the
|
||||
generated registry from their operation request without deferring general
|
||||
module construction until after upstream work.
|
||||
|
||||
Only validation that inherently depends on generated bytes may occur at the
|
||||
handoff. A handoff validation failure is a contextual framework error and fails
|
||||
the run before any consumer in that step begins.
|
||||
|
||||
### Generated artifact handoff
|
||||
|
||||
Only accepted normalized output may cross a step boundary. Raw extraction
|
||||
responses, rejected artifacts, merge intermediates, and validator diagnostics
|
||||
cannot be bound as references.
|
||||
|
||||
The framework serializes the producer through its registered canonical artifact
|
||||
codec and constructs one immutable reference item containing:
|
||||
|
||||
- the declared target slot;
|
||||
- canonical artifact bytes and media type;
|
||||
- artifact kind and schema ID, name, version, and schema digest;
|
||||
- canonical content digest and size; and
|
||||
- producer pipeline, step, lane, and module provenance.
|
||||
|
||||
A generated binding requires exactly one accepted normalized artifact from its
|
||||
producer lane. No artifact is a missing dependency, while more than one is a
|
||||
cardinality error; a typed collection such as an NPC list remains one artifact.
|
||||
Combining several normalized outputs into one reference is aggregation and is
|
||||
outside this scope.
|
||||
|
||||
The existing slot contract remains authoritative for accepted media types,
|
||||
maximum size, and cardinality. Generated content is cloned at ownership
|
||||
boundaries and never exposed through a filesystem path. Manifests and debug
|
||||
summaries record identities and bounded provenance, not artifact content.
|
||||
|
||||
Configuring a generated binding makes that dependency required even when the
|
||||
consumer module declares the underlying slot optional. An accepted artifact
|
||||
whose domain collection is empty is still a valid artifact and may be handed
|
||||
off. If the producer has no accepted normalized artifact, the entire run fails
|
||||
with a deterministic dependency error and no later step begins.
|
||||
|
||||
### Checkpoint reuse and selective recomputation
|
||||
|
||||
Generated references participate in downstream checkpoint dependencies by
|
||||
artifact kind, complete schema identity, media type, and canonical content
|
||||
digest. The pipeline digest protects topology; stage dependency fingerprints
|
||||
protect the exact upstream artifact consumed. The runner must never combine a
|
||||
new or changed producer with stale dependent output.
|
||||
|
||||
Ordinary resume may progressively decode compatible producer and consumer stage
|
||||
checkpoints through the registered codec. Selective recomputation may hydrate a
|
||||
required unselected producer directly from its accepted normalized artifact;
|
||||
its extract and merge state are not prerequisites. Missing, rejected, corrupt,
|
||||
incompatible, or changed accepted state stops the run before dependent
|
||||
execution rather than implicitly rerunning the producer. Independent work
|
||||
remains reusable.
|
||||
|
||||
The operator control `--recompute-step <step-id>` has these semantics:
|
||||
|
||||
- it requires checkpoint recording and `--resume`;
|
||||
- the selected step and all transitive dependents execute rather than reuse
|
||||
their checkpoints;
|
||||
- valid required predecessors and unrelated work remain reusable;
|
||||
- the recompute selection affects loader decisions, not the persistent
|
||||
checkpoint identity of otherwise identical work; and
|
||||
- the command fails before dependent execution if a required predecessor has
|
||||
no reusable accepted artifact.
|
||||
|
||||
Existing `--only` behavior remains unchanged for implicit single-step
|
||||
pipelines. Combining `--only` with explicit multi-step pipelines is outside
|
||||
this scope and should be rejected with actionable guidance rather than given
|
||||
implicit dependency-expansion semantics.
|
||||
|
||||
Checkpoint events, manifests, and diagnostics distinguish executed, reused,
|
||||
forced-recomputed, and dependency-invalidated work. Invalidation reasons are
|
||||
bounded, deterministic, and free of reference content, local paths, or secrets.
|
||||
Old checkpoint state need not be migrated; it must produce a safe, explicit
|
||||
cold miss rather than an error or unsafe reuse.
|
||||
|
||||
### Failure, cancellation, and concurrency
|
||||
|
||||
The existing run-wide worker and provider-call limits apply across every step.
|
||||
Workers may be reused between steps, but concurrency cannot cross a step
|
||||
barrier. A framework error cancels started work using the existing bounded
|
||||
drain behavior and prevents later steps and output encoding. Rejections remain
|
||||
recorded outcomes, but failure to produce a normalized artifact required by a
|
||||
generated binding escalates to the run-level dependency error described above.
|
||||
|
||||
The failed manifest retains completed upstream outcomes, step and lane
|
||||
provenance, rejections, checkpoint events, and the dependency failure without
|
||||
embedding generated artifact content.
|
||||
|
||||
## D&D Proving Workflow
|
||||
|
||||
The production acceptance workflow has two explicit steps:
|
||||
|
||||
1. `identify-npcs` runs the NPC lane through normalization and its complete
|
||||
validator policy.
|
||||
2. `grounded-events` receives the canonical NPC artifact in its step-scoped
|
||||
`npcs` reference and runs spell and combat-turn lanes. The binding reaches
|
||||
spell extraction, combat-turn extraction, and combat-turn normalization.
|
||||
|
||||
Spell and combat-turn lanes may execute concurrently after the handoff. NPC
|
||||
content may ground names and identities but cannot establish a spell cast or
|
||||
combat event; source units remain the only event evidence.
|
||||
|
||||
The maintained workflow uses one ordered-pipeline example instead of a manual
|
||||
two-run NPC-to-spell or NPC-to-combat handoff. Existing module keys, artifact
|
||||
contracts, reference slot names, prompt IDs, and D&D evidence policy remain
|
||||
unchanged.
|
||||
|
||||
## Included Work
|
||||
|
||||
- Configuration parsing, validation, cloning, defaults, redaction, and
|
||||
documentation for explicit steps and structured generated references.
|
||||
- Domain-neutral resolved step, dependency, producer, and consumer identities.
|
||||
- Generated-artifact compatibility metadata on reference-slot contracts,
|
||||
including D&D NPC-list declarations for every `npcs` consumer.
|
||||
- Step-aware preparation metadata and runner orchestration.
|
||||
- Canonical codec handoff into existing reference request contracts.
|
||||
- Required-dependency failure and bounded provenance behavior.
|
||||
- Dependency-aware checkpoint reuse, invalidation, events, and selective step
|
||||
recomputation.
|
||||
- D&D NPC-first production composition for spell and combat-turn consumers.
|
||||
- Refactoring the affected D&D consumers so generated NPC references are
|
||||
available at operation time while retaining early static construction.
|
||||
- Maintained examples and current architecture, configuration, CLI, operations,
|
||||
internal, integration, and testing documentation.
|
||||
- An ADR recording the bounded ordered-step extension to the fixed pipeline
|
||||
architecture and its explicit rejection of a general DAG.
|
||||
|
||||
## Explicitly Out Of Scope
|
||||
|
||||
- D&D item extraction or any other new artifact lane.
|
||||
- Cross-artifact NPC ID fields or artifact-schema migration machinery.
|
||||
- Arbitrary DAGs, conditional branches, loops, joins, dynamic step creation, or
|
||||
module-controlled scheduling.
|
||||
- Multiple source inputs, per-step input adapters, per-step chunk plans, or
|
||||
per-step output encoders.
|
||||
- Aggregating multiple generated artifacts into one reference slot.
|
||||
- Optional or best-effort generated dependencies; a configured dependency is
|
||||
required in this scope.
|
||||
- Prior-run or cross-pipeline generated references.
|
||||
- `--only` dependency closure for explicit multi-step pipelines.
|
||||
- Cross-lane reconciliation or domain concepts in the generic framework.
|
||||
- Live-provider tests or model-quality changes to D&D prompts.
|
||||
|
||||
## Acceptance Criteria
|
||||
|
||||
The scope is complete when:
|
||||
|
||||
- all existing single-step configurations retain their current behavior;
|
||||
- explicit step order and dependency topology resolve deterministically and
|
||||
affect pipeline identity;
|
||||
- invalid producer, consumer, conflict, ordering, type, schema, media, and
|
||||
cardinality configurations fail before source processing when statically
|
||||
discoverable, while content-size violations fail at handoff;
|
||||
- no consumer step begins before all required generated artifacts are accepted,
|
||||
canonicalized, and validated for its target slots;
|
||||
- one producer artifact fans out safely to every compatible target selected by
|
||||
a step-scoped binding;
|
||||
- missing required producer output fails the complete run before dependent work;
|
||||
- changing NPC output invalidates spell and combat-turn checkpoints while
|
||||
leaving compatible independent work reusable;
|
||||
- selective step recomputation executes exactly the selected dependency closure
|
||||
and reports why work was executed, reused, or invalidated;
|
||||
- the D&D ordered workflow supplies NPC content to spell extraction, combat-turn
|
||||
extraction, and combat-turn normalization without treating it as evidence;
|
||||
- completion timing cannot change public ordering, failure selection, or
|
||||
dependency behavior;
|
||||
- output, manifests, checkpoints, and debug artifacts contain the required
|
||||
identities and provenance without leaking generated reference content; and
|
||||
- repository-wide tests, vet, build, maintained-example checks, and
|
||||
documentation validation pass.
|
||||
|
||||
## Testing Strategy
|
||||
|
||||
Tests should protect behavior and invariants rather than the implementation's
|
||||
internal scheduler shape.
|
||||
|
||||
- Configuration contract tests own legacy shorthand, explicit step parsing,
|
||||
source-form discrimination, conflicts, and redaction.
|
||||
- Resolution tests own ordering, global lane uniqueness, dependency validation,
|
||||
slot compatibility, fan-out, cloning, canonical JSON, and digest changes.
|
||||
- Runner tests own step barriers, within-step bounded concurrency, stable
|
||||
ordering, cancellation, required-producer failure, and immutable handoff.
|
||||
- Checkpoint tests own producer decoding, exact dependency matching, transitive
|
||||
invalidation, forced recomputation, cold misses, and bounded decisions.
|
||||
- One CLI contract test should cover the recompute control and its invalid
|
||||
combinations.
|
||||
- One D&D integration test with offline fake LLM responses should prove the
|
||||
complete NPC-to-spell-and-combat handoff, including combat normalization.
|
||||
- Maintained configuration examples should be parsed and resolved through the
|
||||
production catalog.
|
||||
|
||||
Do not add scheduler choreography tests, exact goroutine-count assertions,
|
||||
complete manifest snapshots, exact diagnostic strings, or duplicated tests for
|
||||
every invalid configuration at every layer. No test may require credentials or
|
||||
a live model provider.
|
||||
|
||||
## Open Questions
|
||||
|
||||
None required to define this scope. Exact internal type names and implementation
|
||||
decomposition are intentionally not feature-policy decisions.
|
||||
318
docs/roadmap/promptkit.md
Normal file
318
docs/roadmap/promptkit.md
Normal file
@@ -0,0 +1,318 @@
|
||||
# PromptKit v0.5 Integration And LLM Profile Policy
|
||||
|
||||
## Purpose
|
||||
|
||||
This roadmap defines the target state for upgrading Notarius from PromptKit
|
||||
v0.3.0 to v0.5.0 and adopting the upstream runtime and profile facilities that
|
||||
directly improve Notarius. It also defines the application policy for stable,
|
||||
domain-oriented LLM profile names, operator overrides, pipeline inheritance,
|
||||
profile validation, provider defaults, checkpoint identity, and documentation.
|
||||
|
||||
The ordered work needed to reach this state belongs in
|
||||
[the implementation plan](implementation.md). Current behavior remains defined
|
||||
by the canonical documentation outside `docs/roadmap/` until the corresponding
|
||||
work is implemented.
|
||||
|
||||
## Background
|
||||
|
||||
Notarius currently pins PromptKit v0.3.0. Its adapter prepares a request once
|
||||
for debug material and then independently runs the original request, causing
|
||||
PromptKit to prepare the same logical call a second time. The CLI validates an
|
||||
explicit profile by preparing a synthetic prompt. PromptKit profile selection
|
||||
can be repeated on individual module bindings or replaced for one invocation
|
||||
with `--llm-profile`, but a configured pipeline cannot yet declare one inherited
|
||||
profile policy.
|
||||
|
||||
PromptKit v0.4.0 and v0.5.0 add the upstream boundaries needed to improve these
|
||||
areas:
|
||||
|
||||
- [v0.4.0](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0/docs/releases/v0.4.0.md)
|
||||
adds opaque prepared executions, exact profile and prompt inspection, and a
|
||||
typed backend-capacity error;
|
||||
- [v0.5.0](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0/docs/releases/v0.5.0.md)
|
||||
adds application fallback profile filesystems and stops sending unset
|
||||
optional sampling controls as framework-selected provider values; and
|
||||
- the [v0.5.0 format contract](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0/docs/formats.md)
|
||||
defines the resulting profile-source and execution-setting precedence.
|
||||
|
||||
A source-compatibility test of the current Notarius repository against
|
||||
PromptKit v0.5.0 completed successfully. The work is therefore primarily an
|
||||
intentional runtime and configuration migration rather than a repair for a
|
||||
breaking Go API change.
|
||||
|
||||
## Goals
|
||||
|
||||
- Pin and document PromptKit v0.5.0 as Notarius's supported upstream contract.
|
||||
- Execute the exact prepared request snapshot whose safe details are recorded
|
||||
in Notarius debug material.
|
||||
- Validate configured PromptKit profiles through the upstream inspection API
|
||||
without synthetic prompts, provider calls, or credential-value access.
|
||||
- Give Notarius an application-owned, operator-overridable
|
||||
`dnd-extraction` profile fallback.
|
||||
- Let a pipeline choose one default LLM profile without repeating that ID on
|
||||
every LLM-backed binding.
|
||||
- Apply profile inheritance and run-wide overrides only where the resolved
|
||||
module or validator can use an LLM.
|
||||
- Preserve accurate checkpoint invalidation, effective profile provenance,
|
||||
redaction, cancellation, concurrency, and provider-neutral module contracts.
|
||||
- Provide operators with one clear deployment pattern for production,
|
||||
development, and local profile definitions.
|
||||
|
||||
## Target End State
|
||||
|
||||
### PromptKit Runtime Boundary
|
||||
|
||||
Notarius depends on PromptKit v0.5.0 and uses its public APIs rather than
|
||||
reimplementing source or execution resolution.
|
||||
|
||||
For each structured completion, the adapter:
|
||||
|
||||
1. builds one PromptKit run request from the provider-neutral Notarius request;
|
||||
2. calls `PrepareExecution` once;
|
||||
3. immediately arranges an idempotent `Discard` for every unexecuted handle;
|
||||
4. obtains credential-redacted `Details` for debug and response metadata; and
|
||||
5. calls `RunPrepared` so generation uses that exact frozen snapshot.
|
||||
|
||||
The debug prompt and successful result therefore describe the same selected
|
||||
profile, rendered messages, input bytes, session, output contract, and effective
|
||||
settings even when a filesystem-backed source changes concurrently. PromptKit
|
||||
handle types remain private to `internal/framework/llm`.
|
||||
|
||||
PromptKit admission failures continue to match Notarius's provider-neutral
|
||||
`ErrLLMCapacityExceeded` contract. When PromptKit supplies a `CapacityError`,
|
||||
the adapter obtains the normalized backend ID through `errors.As` and may add it
|
||||
to safe application-owned diagnostics without parsing upstream error wording.
|
||||
The backend ID does not become a provider-specific module contract.
|
||||
|
||||
### Optional Provider Controls
|
||||
|
||||
Notarius accepts PromptKit v0.5.0's new behavior for `temperature`,
|
||||
`max_tokens`, and `top_p`: an unset setting is omitted from compatible provider
|
||||
requests and the provider chooses its own default. Notarius does not restore
|
||||
PromptKit's former implicit `top_p: 1` value globally.
|
||||
|
||||
An operator who requires a particular value specifies it in the selected
|
||||
PromptKit profile. The application fallback described below intentionally
|
||||
leaves these controls unset. A human-reviewed D&D extraction comparison should
|
||||
be performed after the upgrade, but paid or nondeterministic model output is
|
||||
not part of the default automated test suite.
|
||||
|
||||
### Profile Inspection
|
||||
|
||||
Pipeline-aware configuration validation uses `Engine.InspectProfile` for every
|
||||
effective explicit profile ID. It verifies that the profile exists, parses and
|
||||
validates, resolves its backend and target, and is compatible with the engine's
|
||||
registered backends. It does not create a synthetic prompt, load prompt inputs,
|
||||
contact a provider, or require credential values to exist in the validation
|
||||
process environment.
|
||||
|
||||
Credential availability is execution-time state. PromptKit preparation still
|
||||
enforces the selected profile's credential contract before generation. This
|
||||
keeps `notarius config validate` useful in build and deployment validation
|
||||
environments where secrets are deliberately absent.
|
||||
|
||||
PromptKit construction for inspection and execution uses one shared internal
|
||||
profile-source and backend-option path. The CLI does not expose PromptKit public
|
||||
types across the Notarius LLM boundary merely to perform inspection.
|
||||
|
||||
`InspectPrompt` is not adopted merely because it exists. It remains available
|
||||
for a later, separately defined module-to-prompt interface preflight if a
|
||||
concrete validation requirement justifies that additional contract.
|
||||
|
||||
### Application And Operator Profile Sources
|
||||
|
||||
Notarius embeds one ordinary PromptKit YAML profile with the stable ID
|
||||
`dnd-extraction`. It is an application fallback registered through
|
||||
`WithFallbackProfileFS`, is owned by the D&D module family, and initially
|
||||
preserves the current effective D&D baseline:
|
||||
|
||||
- backend: PromptKit's built-in `openrouter` backend;
|
||||
- model: `openai/gpt-5.6-luna`;
|
||||
- reasoning effort: unset, allowing OpenAI's backend to apply its default of
|
||||
`medium`;
|
||||
- generation timeout: 240 seconds;
|
||||
- service tier: `flex`; and
|
||||
- no application-selected `temperature`, `max_tokens`, or `top_p`.
|
||||
|
||||
All maintained D&D LLM prompt definitions use `dnd-extraction` as their
|
||||
`default_profile`. The ID communicates workload intent rather than a provider,
|
||||
model, or environment. Changing the embedded fallback is an intentional
|
||||
Notarius execution-policy change and participates in checkpoint identity.
|
||||
|
||||
Effective profile definitions resolve in PromptKit's order:
|
||||
|
||||
1. programmatic in-memory profiles used by tests or explicit consumers;
|
||||
2. the operator source configured by `promptkit.profile_file` or
|
||||
`promptkit.profile_dir`;
|
||||
3. the Notarius application fallback source; and
|
||||
4. PromptKit's embedded built-in catalog.
|
||||
|
||||
Only an absent ID falls through to the next source. A matching profile is a
|
||||
complete definition: fields are not merged with a lower-precedence definition,
|
||||
and a malformed matching operator profile fails rather than silently selecting
|
||||
the application fallback.
|
||||
|
||||
Production, development, and local deployments should normally provide
|
||||
different complete definitions for the same `dnd-extraction` ID. An operator
|
||||
source is optional because the application fallback keeps the maintained D&D
|
||||
workflow usable, but a deployment that needs an intentional model or backend
|
||||
policy should configure its own definition.
|
||||
|
||||
### Domain Ownership And Asset Assembly
|
||||
|
||||
The D&D fallback profile remains under `internal/modules/dnd` and is registered
|
||||
by the D&D registrar, consistent with ADR-0004. Generic LLM plumbing knows how
|
||||
to collect and flatten application fallback profile filesystems but contains no
|
||||
D&D model or policy knowledge.
|
||||
|
||||
The shared asset registry detects invalid roots, unreadable sources, and
|
||||
duplicate flattened paths. PromptKit remains responsible for strict profile
|
||||
YAML parsing, duplicate profile-ID detection, source precedence, and effective
|
||||
target resolution. The same assembled fallback source is supplied to runtime
|
||||
execution and CLI profile inspection.
|
||||
|
||||
### Explicit Module Execution Metadata
|
||||
|
||||
Every registered input, chunk, extract, merge, normalize, and output module
|
||||
declares one required execution class: `deterministic` or `llm_backed`.
|
||||
Validator registrations continue to declare the same distinction through their
|
||||
validator specifications.
|
||||
|
||||
The registered specification is authoritative for configuration resolution.
|
||||
Current production classifications are:
|
||||
|
||||
- the D&D scene chunker, all D&D extractors, and the D&D NPC normalizer are
|
||||
LLM-backed;
|
||||
- the Seriatim input adapter, generic chunker, all current mergers, all other
|
||||
current normalizers, and the JSON output encoder are deterministic; and
|
||||
- current validators retain their declared classifications.
|
||||
|
||||
Missing or unsupported execution metadata is a registration error. Explicitly
|
||||
assigning `llm_profile` to a deterministic module or validator is a pipeline
|
||||
resolution error. The framework does not infer execution class by inspecting
|
||||
domain package names or concrete implementation types at runtime.
|
||||
|
||||
The module specification replaces the chunk runner's special runtime
|
||||
execution-class probe. Effective resolved bindings already express the result:
|
||||
only LLM-backed bindings may retain a non-empty profile.
|
||||
|
||||
### Pipeline-Level Profile Default
|
||||
|
||||
Configuration version 4 gains one optional non-empty pipeline field:
|
||||
|
||||
```yaml
|
||||
pipelines:
|
||||
dnd-session:
|
||||
llm_profile: dnd-extraction
|
||||
```
|
||||
|
||||
No configuration-version increment is required because the field is additive
|
||||
and existing files remain valid. An explicitly present blank value is invalid.
|
||||
|
||||
For every selected LLM-backed module and validator, the effective profile uses
|
||||
this precedence:
|
||||
|
||||
1. non-empty run-wide `--llm-profile` override;
|
||||
2. binding-specific `llm_profile`;
|
||||
3. pipeline-level `llm_profile`; and
|
||||
4. the prompt definition's `default_profile`, represented by an empty effective
|
||||
Notarius binding profile.
|
||||
|
||||
The run-wide override and inherited pipeline default never attach to a
|
||||
deterministic binding. Binding-specific exceptions remain available when one
|
||||
operation needs a different cost, latency, quality, backend, or reasoning
|
||||
policy.
|
||||
|
||||
Inheritance is resolved after module and validator selection, including
|
||||
`--only` lane filtering, but before effective-pipeline validation, digest
|
||||
construction, explicit-profile inspection, checkpoint construction,
|
||||
preparation, execution, or provenance capture. Only profiles used by selected
|
||||
LLM-backed bindings are inspected. An unused pipeline default in a pipeline
|
||||
with no selected LLM-backed work does not require an otherwise unused profile
|
||||
to exist.
|
||||
|
||||
The resolved pipeline contains effective binding profiles rather than a second
|
||||
runtime inheritance mechanism. Two pipelines that differ only by spelling the
|
||||
same effective policy once as a pipeline default and once on every LLM-backed
|
||||
binding have the same semantic resolved digest. Changing an effective profile
|
||||
changes the digest and applicable checkpoint identity.
|
||||
|
||||
### Provenance And Checkpoints
|
||||
|
||||
The PromptKit profile-source checkpoint fingerprint covers:
|
||||
|
||||
- the PromptKit v0.5.0 built-in profile catalog identity;
|
||||
- exact application fallback profile asset content; and
|
||||
- exact configured operator profile YAML content, when present.
|
||||
|
||||
The existing local-backend target fingerprint remains separate and continues
|
||||
to exclude scheduling-only concurrency limits. Fingerprints contain hashes and
|
||||
stable markers, not profile contents, filesystem paths, endpoints, credentials,
|
||||
or other secrets.
|
||||
|
||||
Changing the PromptKit version, application fallback, operator profile, or
|
||||
effective pipeline profile makes incompatible LLM checkpoints ineligible for
|
||||
reuse. The dependency upgrade is expected to invalidate checkpoints produced
|
||||
under v0.3.0.
|
||||
|
||||
Successful run manifests continue to record only profiles actually selected by
|
||||
PromptKit, including their effective model, backend, and reasoning metadata.
|
||||
Debug output reports the same effective execution snapshot used for generation.
|
||||
|
||||
### Operator Documentation And Examples
|
||||
|
||||
Canonical documentation clearly distinguishes:
|
||||
|
||||
- Notarius prompt and schema assets embedded in the application;
|
||||
- Notarius application fallback profiles embedded in the application;
|
||||
- PromptKit's own embedded built-in profiles; and
|
||||
- operator profile files on the deployment filesystem.
|
||||
|
||||
The configuration reference owns the pipeline field, profile-source fields,
|
||||
validation rules, and precedence. Operations owns deployment layout, working
|
||||
directory behavior, credentials, and environment-specific profile management.
|
||||
The PromptKit integration document owns the pinned upstream contract and
|
||||
source-precedence boundary. Internal documents describe asset registration,
|
||||
resolution, inspection, prepared execution, fingerprinting, and tests without
|
||||
duplicating user-facing field definitions.
|
||||
|
||||
The maintained examples continue to include only the minimal and complete D&D
|
||||
configurations. They use the stable `dnd-extraction` policy, and one maintained
|
||||
PromptKit profile file under `examples/` demonstrates an operator override.
|
||||
Examples remain secret-free and are validated without live provider calls.
|
||||
|
||||
## Out Of Scope
|
||||
|
||||
- Implementing the separate deterministic prompt-session identity roadmap
|
||||
item.
|
||||
- Changing the default `concurrency.total_llm` value; PromptKit's retained
|
||||
OpenRouter capacity of 16 remains relevant to that separate item.
|
||||
- Adding model evaluation as a deterministic or CI correctness gate.
|
||||
- Automatically selecting production, development, or local environments.
|
||||
Deployment configuration chooses the operator profile source.
|
||||
- Profile inheritance, partial profile merging, or cross-profile aliases.
|
||||
- Exposing PromptKit types to modules, validators, durable output contracts, or
|
||||
public configuration structures.
|
||||
- Adopting `InspectPrompt` without a separately justified prompt-interface
|
||||
validation contract.
|
||||
|
||||
## Acceptance Criteria
|
||||
|
||||
- Notarius builds and its offline test suite passes with PromptKit v0.5.0.
|
||||
- Every structured completion executes the exact snapshot used for safe debug
|
||||
prompt details.
|
||||
- Profile preflight uses profile inspection and no synthetic prompt.
|
||||
- The embedded `dnd-extraction` fallback resolves without an operator source,
|
||||
and a matching valid operator profile replaces it completely.
|
||||
- Every production module has explicit, correct execution metadata.
|
||||
- Pipeline, binding, CLI, and prompt-default precedence behaves as defined for
|
||||
modules and validators, while deterministic bindings remain profile-free.
|
||||
- Effective profiles participate in pipeline digests, profile inspection,
|
||||
checkpoint identity, debug records, and run provenance at the appropriate
|
||||
boundaries.
|
||||
- The dependency and application fallback changes invalidate incompatible old
|
||||
checkpoints without exposing profile or credential content.
|
||||
- Canonical documentation and maintained examples accurately describe and
|
||||
exercise the implemented operator workflow.
|
||||
- Default tests remain deterministic, offline, credential-free, and focused on
|
||||
Notarius-owned behavior rather than duplicating PromptKit's upstream suite.
|
||||
@@ -1,22 +0,0 @@
|
||||
version: 3
|
||||
output:
|
||||
directory: ./notarius-output
|
||||
cache:
|
||||
chunk_plans:
|
||||
mode: bypass
|
||||
directory: ./notarius-cache/chunk-plans
|
||||
checkpoints:
|
||||
enabled: false
|
||||
directory: ./notarius-cache/checkpoints
|
||||
debug:
|
||||
directory: ./notarius-debug
|
||||
pipelines:
|
||||
dnd-combat:
|
||||
input: seriatim
|
||||
chunk: generic
|
||||
artifacts:
|
||||
combat:
|
||||
extract:
|
||||
module: dnd/combat-turns
|
||||
retries: 2
|
||||
normalize: dnd/combat-turns
|
||||
85
examples/dnd-complete-transcript.json
Normal file
85
examples/dnd-complete-transcript.json
Normal file
@@ -0,0 +1,85 @@
|
||||
{
|
||||
"metadata": {
|
||||
"id": "session-ravenfall",
|
||||
"title": "The Ravenfall Watchtower"
|
||||
},
|
||||
"segments": [
|
||||
{
|
||||
"id": 1,
|
||||
"start": 0,
|
||||
"end": 14,
|
||||
"speaker": "DM",
|
||||
"text": "Recap: last session, the party learned that Elder Rowan vanished near the Ravenfall watchtower."
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"start": 14,
|
||||
"end": 25,
|
||||
"speaker": "Player",
|
||||
"text": "Out of character, we agree to investigate the watchtower before the next game."
|
||||
},
|
||||
{
|
||||
"id": 3,
|
||||
"start": 25,
|
||||
"end": 39,
|
||||
"speaker": "DM",
|
||||
"text": "Aria and Borin arrive at the ruined Ravenfall watchtower as dusk settles over the road."
|
||||
},
|
||||
{
|
||||
"id": 4,
|
||||
"start": 39,
|
||||
"end": 55,
|
||||
"speaker": "Mira Thorn",
|
||||
"text": "Mira Thorn steps from the doorway and says, \"Elder Rowan warned me that Kesh would return for the relic.\""
|
||||
},
|
||||
{
|
||||
"id": 5,
|
||||
"start": 55,
|
||||
"end": 70,
|
||||
"speaker": "DM",
|
||||
"text": "Mira leads the party to a hidden cache. The party discovers a moonblade and acquires 20 silver pieces."
|
||||
},
|
||||
{
|
||||
"id": 6,
|
||||
"start": 70,
|
||||
"end": 83,
|
||||
"speaker": "Aria",
|
||||
"text": "Aria hands her healing potion to Borin so he can carry it into the tower."
|
||||
},
|
||||
{
|
||||
"id": 7,
|
||||
"start": 83,
|
||||
"end": 96,
|
||||
"speaker": "DM",
|
||||
"text": "Kesh, the goblin captain, orders the raiders to attack. Roll initiative."
|
||||
},
|
||||
{
|
||||
"id": 8,
|
||||
"start": 96,
|
||||
"end": 110,
|
||||
"speaker": "DM",
|
||||
"text": "On Kesh's turn, he strikes Borin with his scimitar. Borin drinks the healing potion on his turn."
|
||||
},
|
||||
{
|
||||
"id": 9,
|
||||
"start": 110,
|
||||
"end": 124,
|
||||
"speaker": "Aria",
|
||||
"text": "Aria casts Cure Wounds on Borin, then invokes Aegis of Emberfall as Kesh closes in."
|
||||
},
|
||||
{
|
||||
"id": 10,
|
||||
"start": 124,
|
||||
"end": 137,
|
||||
"speaker": "DM",
|
||||
"text": "Kesh casts Shield as a reaction against Borin's counterattack, but the party drives the raiders away."
|
||||
},
|
||||
{
|
||||
"id": 11,
|
||||
"start": 137,
|
||||
"end": 150,
|
||||
"speaker": "DM",
|
||||
"text": "After the battle, Aria pays 5 silver pieces to repair the watchtower gate."
|
||||
}
|
||||
]
|
||||
}
|
||||
103
examples/dnd-complete.config.yml
Normal file
103
examples/dnd-complete.config.yml
Normal file
@@ -0,0 +1,103 @@
|
||||
version: 4
|
||||
promptkit:
|
||||
profile_file: ./examples/profiles/dnd-extraction.yml
|
||||
concurrency:
|
||||
total_llm: 2
|
||||
stage_workers:
|
||||
extract: 2
|
||||
output:
|
||||
directory: ./notarius-output
|
||||
cache:
|
||||
chunk_plans:
|
||||
mode: auto
|
||||
directory: ./notarius-cache/chunk-plans
|
||||
checkpoints:
|
||||
enabled: true
|
||||
directory: ./notarius-cache/checkpoints
|
||||
debug:
|
||||
directory: ./notarius-debug
|
||||
pipelines:
|
||||
dnd-session:
|
||||
llm_profile: dnd-extraction
|
||||
input: seriatim
|
||||
# Stable campaign context is shared by every module that accepts these slots.
|
||||
references:
|
||||
party: ./dnd-party.txt
|
||||
glossary: ./dnd-glossary.txt
|
||||
chunk:
|
||||
module: dnd/scenes
|
||||
retries: 2
|
||||
output:
|
||||
module: json
|
||||
options:
|
||||
include_chunk_map: true
|
||||
evidence_context:
|
||||
enabled: true
|
||||
window_units: 3
|
||||
lanes:
|
||||
- item-events
|
||||
- npcs
|
||||
- spells
|
||||
- combat-turns
|
||||
- npc-interactions
|
||||
steps:
|
||||
# Establish session-wide reference artifacts alongside independent item events.
|
||||
- id: describe-session
|
||||
artifacts:
|
||||
item-events:
|
||||
extract:
|
||||
module: dnd/item-events
|
||||
retries: 2
|
||||
merge: appendorder
|
||||
normalize: dnd/item-events
|
||||
npcs:
|
||||
extract:
|
||||
module: dnd/npcs
|
||||
retries: 2
|
||||
merge: appendorder
|
||||
normalize:
|
||||
module: dnd/npcs
|
||||
retries: 2
|
||||
scene-descriptions:
|
||||
extract:
|
||||
module: dnd/scene-descriptions
|
||||
retries: 2
|
||||
merge: appendorder
|
||||
normalize: dnd/scene-descriptions
|
||||
- id: extract-events
|
||||
# Accepted NPC grounding and scene-description eligibility artifacts are
|
||||
# supplied in memory to their compatible consumers in this step.
|
||||
references:
|
||||
npcs:
|
||||
artifact:
|
||||
step: describe-session
|
||||
lane: npcs
|
||||
scene_descriptions:
|
||||
artifact:
|
||||
step: describe-session
|
||||
lane: scene-descriptions
|
||||
artifacts:
|
||||
spells:
|
||||
extract:
|
||||
module: dnd/spells
|
||||
retries: 2
|
||||
references:
|
||||
spell_catalog: ./dnd-spell-catalog.json
|
||||
merge: appendorder
|
||||
# Stage-local file references are intentionally bound at each stage.
|
||||
normalize:
|
||||
module: dnd/spells
|
||||
references:
|
||||
spell_catalog: ./dnd-spell-catalog.json
|
||||
combat-turns:
|
||||
extract:
|
||||
module: dnd/combat-turns
|
||||
retries: 2
|
||||
merge: appendorder
|
||||
normalize: dnd/combat-turns
|
||||
npc-interactions:
|
||||
extract:
|
||||
module: dnd/npc-interactions
|
||||
retries: 2
|
||||
merge: appendorder
|
||||
normalize: dnd/npc-interactions
|
||||
8
examples/dnd-glossary.txt
Normal file
8
examples/dnd-glossary.txt
Normal file
@@ -0,0 +1,8 @@
|
||||
Ravenfall watchtower: a ruined watchtower near the party's current route.
|
||||
Mira Thorn: the watchtower's keeper.
|
||||
Elder Rowan: a missing local scholar.
|
||||
Kesh: a goblin captain leading raiders.
|
||||
Moonblade: a blade found in the watchtower's hidden cache.
|
||||
Cure Wounds: a healing spell.
|
||||
Shield: a defensive reaction spell.
|
||||
Aegis of Emberfall: a campaign spell recorded in the supplied catalog overlay.
|
||||
9
examples/dnd-minimal.config.yml
Normal file
9
examples/dnd-minimal.config.yml
Normal file
@@ -0,0 +1,9 @@
|
||||
version: 4
|
||||
pipelines:
|
||||
dnd-session:
|
||||
llm_profile: dnd-extraction
|
||||
input: seriatim
|
||||
artifacts:
|
||||
spells:
|
||||
extract: dnd/spells
|
||||
normalize: dnd/spells
|
||||
@@ -1,37 +0,0 @@
|
||||
version: 3
|
||||
output:
|
||||
directory: ./notarius-output
|
||||
cache:
|
||||
chunk_plans:
|
||||
mode: bypass
|
||||
checkpoints:
|
||||
enabled: false
|
||||
directory: ""
|
||||
debug:
|
||||
directory: ./notarius-debug
|
||||
pipelines:
|
||||
dnd-npc-grounded:
|
||||
input: seriatim
|
||||
steps:
|
||||
- id: identify-npcs
|
||||
artifacts:
|
||||
npcs:
|
||||
extract:
|
||||
module: dnd/npcs
|
||||
retries: 2
|
||||
normalize: dnd/npcs
|
||||
- id: grounded-events
|
||||
references:
|
||||
npcs:
|
||||
artifact:
|
||||
step: identify-npcs
|
||||
lane: npcs
|
||||
artifacts:
|
||||
spells:
|
||||
extract: dnd/spells
|
||||
normalize: dnd/spells
|
||||
combat:
|
||||
extract:
|
||||
module: dnd/combat-turns
|
||||
retries: 2
|
||||
normalize: dnd/combat-turns
|
||||
@@ -1,21 +0,0 @@
|
||||
version: 3
|
||||
output:
|
||||
directory: ./notarius-output
|
||||
cache:
|
||||
chunk_plans:
|
||||
mode: bypass
|
||||
checkpoints:
|
||||
enabled: false
|
||||
directory: ""
|
||||
debug:
|
||||
directory: ./notarius-debug
|
||||
pipelines:
|
||||
dnd-session:
|
||||
input: seriatim
|
||||
chunk: generic
|
||||
artifacts:
|
||||
npcs:
|
||||
extract:
|
||||
module: dnd/npcs
|
||||
retries: 2
|
||||
normalize: dnd/npcs
|
||||
@@ -1,3 +1,2 @@
|
||||
Aria: party cleric and recurring healer.
|
||||
Borin: fighter ally.
|
||||
Bandit mage: hostile spellcaster.
|
||||
@@ -1,2 +0,0 @@
|
||||
Cure Wounds: healing spell cast by touch.
|
||||
Shield: defensive reaction spell.
|
||||
@@ -1,39 +0,0 @@
|
||||
version: 3
|
||||
concurrency:
|
||||
total_llm: 1
|
||||
stage_workers:
|
||||
extract: 1
|
||||
output:
|
||||
directory: ./notarius-output
|
||||
cache:
|
||||
chunk_plans:
|
||||
directory: /var/cache/notarius/chunk-plans
|
||||
mode: auto
|
||||
checkpoints:
|
||||
enabled: false
|
||||
directory: /var/cache/notarius/checkpoints
|
||||
debug:
|
||||
directory: ./notarius-debug
|
||||
pipelines:
|
||||
dnd-session:
|
||||
input: seriatim
|
||||
references:
|
||||
party: ./dnd-spells-roster.txt
|
||||
glossary: ./dnd-spells-glossary.txt
|
||||
chunk:
|
||||
module: generic
|
||||
options:
|
||||
max_units: 50
|
||||
artifacts:
|
||||
spells:
|
||||
extract:
|
||||
module: dnd/spells
|
||||
retries: 2
|
||||
# Overlay behavior binds the same catalog independently at each stage.
|
||||
references:
|
||||
spell_catalog: ./dnd-spells-catalog.json
|
||||
normalize:
|
||||
module: dnd/spells
|
||||
# Normalize-stage references are local and must be bound explicitly.
|
||||
references:
|
||||
spell_catalog: ./dnd-spells-catalog.json
|
||||
@@ -1,19 +0,0 @@
|
||||
version: 3
|
||||
output:
|
||||
directory: ./notarius-output
|
||||
cache:
|
||||
chunk_plans:
|
||||
mode: bypass
|
||||
checkpoints:
|
||||
enabled: false
|
||||
directory: ""
|
||||
debug:
|
||||
directory: ./notarius-debug
|
||||
pipelines:
|
||||
dnd-session:
|
||||
input: seriatim
|
||||
artifacts:
|
||||
spells:
|
||||
extract: dnd/spells
|
||||
# Base-only behavior: normalization uses the embedded SRD catalog.
|
||||
normalize: dnd/spells
|
||||
5
examples/profiles/dnd-extraction.yml
Normal file
5
examples/profiles/dnd-extraction.yml
Normal file
@@ -0,0 +1,5 @@
|
||||
id: dnd-extraction
|
||||
backend: openrouter
|
||||
model: openai/gpt-5.6-luna
|
||||
timeout_seconds: 240
|
||||
service_tier: flex
|
||||
2
go.mod
2
go.mod
@@ -3,7 +3,7 @@ module gitea.maximumdirect.net/eric/notarius
|
||||
go 1.25.5
|
||||
|
||||
require (
|
||||
gitea.maximumdirect.net/eric/scriptorium v0.11.1
|
||||
gitea.maximumdirect.net/eric/promptkit v0.5.0
|
||||
github.com/santhosh-tekuri/jsonschema/v6 v6.0.2
|
||||
gopkg.in/yaml.v3 v3.0.1
|
||||
)
|
||||
|
||||
8
go.sum
8
go.sum
@@ -1,13 +1,9 @@
|
||||
gitea.maximumdirect.net/eric/scriptorium v0.11.0 h1:rjvbt9FTaWHxYlHq7QlUzmMVUt3QdbTmeCkmH81N//o=
|
||||
gitea.maximumdirect.net/eric/scriptorium v0.11.0/go.mod h1:FQ5lEuNxmrQyNgIomkpZdxvfTC0jWjbXYuq3tbJWF64=
|
||||
gitea.maximumdirect.net/eric/scriptorium v0.11.1 h1:zBKtB3+fP8FcHGI8DJD99CiTL6crAGitBhWtE+xYJHc=
|
||||
gitea.maximumdirect.net/eric/scriptorium v0.11.1/go.mod h1:FQ5lEuNxmrQyNgIomkpZdxvfTC0jWjbXYuq3tbJWF64=
|
||||
gitea.maximumdirect.net/eric/promptkit v0.5.0 h1:jnpazLyyNhWrB2xzwwtUkNUfktkTdkENTwuSPnKiYrc=
|
||||
gitea.maximumdirect.net/eric/promptkit v0.5.0/go.mod h1:R95NM6fbMDGDC0/UomgnSBP6ui2ns+8SZb8bESNvrDQ=
|
||||
github.com/dlclark/regexp2 v1.11.0 h1:G/nrcoOa7ZXlpoa/91N3X7mM3r8eIlMBBJZvsz/mxKI=
|
||||
github.com/dlclark/regexp2 v1.11.0/go.mod h1:DHkYz0B9wPfa6wondMfaivmHpzrQ3v9q8cnmRbL6yW8=
|
||||
github.com/santhosh-tekuri/jsonschema/v6 v6.0.2 h1:KRzFb2m7YtdldCEkzs6KqmJw4nqEVZGK7IN2kJkjTuQ=
|
||||
github.com/santhosh-tekuri/jsonschema/v6 v6.0.2/go.mod h1:JXeL+ps8p7/KNMjDQk3TCwPpBy0wYklyWTfbkIzdIFU=
|
||||
golang.org/x/text v0.14.0 h1:ScX5w1eTa3QqT8oi6+ziP7dTV1S2+ALU0bI+0zXKWiQ=
|
||||
golang.org/x/text v0.14.0/go.mod h1:18ZOQIKpY8NJVqYksKHtTdi31H5itFRjB5/qKTNYzSU=
|
||||
golang.org/x/text v0.40.0 h1:Ub2Z6/xjgF1WrYQz2nuITOEegKFtiIy+rieRJ5lHZKs=
|
||||
golang.org/x/text v0.40.0/go.mod h1:hpnzDAfGV753zIKo+wk3u1bVKCGPbrnF7+7LBF/UHVY=
|
||||
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM=
|
||||
|
||||
@@ -57,8 +57,8 @@ func TestAssembledSpellPipelineNormalizesMergedCasts(t *testing.T) {
|
||||
t.Fatalf("normalized casts = %#v, want collapsed duplicate plus distinct evidence", normalized.SpellCasts)
|
||||
}
|
||||
first, distinct := normalized.SpellCasts[0], normalized.SpellCasts[1]
|
||||
if first.Spell != "Cure Wounds" || first.Caster != " Aria \t" || first.Effect != "first occurrence" || first.NarrativeDescription != "first narrative" {
|
||||
t.Fatalf("retained cast = %#v, want canonical spell with first occurrence fields", first)
|
||||
if first.Spell != "Cure Wounds" || first.Caster != " Aria \t" {
|
||||
t.Fatalf("retained cast = %#v, want canonical spell with first occurrence caster", first)
|
||||
}
|
||||
if !reflect.DeepEqual(first.SourceRefs, []source.SourceRef{{SourceID: "session-alpha", StartUnitID: 1, EndUnitID: 1}, {SourceID: "session-alpha", StartUnitID: 2, EndUnitID: 2}}) {
|
||||
t.Fatalf("retained refs = %#v, want sorted complete evidence", first.SourceRefs)
|
||||
@@ -215,6 +215,7 @@ func assembledSpellPipeline(t *testing.T, options assembledSpellPipelineOptions)
|
||||
if err := pipeline.RegisterExtractor[dnd.SpellList](components.registries.Extractors, pipeline.ModuleSpec{
|
||||
Key: assembledSpellExtractorKey,
|
||||
Stage: pipeline.StageExtract,
|
||||
ExecutionClass: contracts.ExecutionClassDeterministic,
|
||||
Requires: []string{"chunks", "source.transcript"},
|
||||
Provides: []string{"dnd.spell_casts"},
|
||||
ArtifactKind: dnd.SpellListKind,
|
||||
@@ -271,7 +272,7 @@ func (e *assembledSpellExtractor) Extract(ctx context.Context, req contracts.Typ
|
||||
if e.unknownSpell {
|
||||
if req.Chunk.Index == 0 {
|
||||
return contracts.TypedExtractionResult[dnd.SpellList]{Value: dnd.SpellList{SpellCasts: []dnd.SpellCast{{
|
||||
Caster: "Aria", Spell: "Mysterious Burst", Effect: "an unknown magical effect", NarrativeDescription: "Aria produces a mysterious burst.", SourceRefs: []source.SourceRef{refOne},
|
||||
Caster: "Aria", Spell: "Mysterious Burst", SourceRefs: []source.SourceRef{refOne},
|
||||
}}}}, nil
|
||||
}
|
||||
return contracts.TypedExtractionResult[dnd.SpellList]{Value: dnd.SpellList{SpellCasts: []dnd.SpellCast{}}}, nil
|
||||
@@ -279,12 +280,12 @@ func (e *assembledSpellExtractor) Extract(ctx context.Context, req contracts.Typ
|
||||
switch req.Chunk.Index {
|
||||
case 0:
|
||||
return contracts.TypedExtractionResult[dnd.SpellList]{Value: dnd.SpellList{SpellCasts: []dnd.SpellCast{{
|
||||
Caster: " Aria \t", Spell: " cure wounds ", Effect: "first occurrence", NarrativeDescription: "first narrative", SourceRefs: []source.SourceRef{refTwo, refOne},
|
||||
Caster: " Aria \t", Spell: " cure wounds ", SourceRefs: []source.SourceRef{refTwo, refOne},
|
||||
}}}}, nil
|
||||
case 1:
|
||||
return contracts.TypedExtractionResult[dnd.SpellList]{Value: dnd.SpellList{SpellCasts: []dnd.SpellCast{
|
||||
{Caster: "aria", Spell: "Cure Wounds", Effect: "removed occurrence", NarrativeDescription: "removed narrative", SourceRefs: []source.SourceRef{refOne, refTwo}},
|
||||
{Caster: "aria", Spell: "Cure Wounds", Effect: "different evidence", NarrativeDescription: "different narrative", SourceRefs: []source.SourceRef{refTwo}},
|
||||
{Caster: "aria", Spell: "Cure Wounds", SourceRefs: []source.SourceRef{refOne, refTwo}},
|
||||
{Caster: "aria", Spell: "Cure Wounds", SourceRefs: []source.SourceRef{refTwo}},
|
||||
}}}, nil
|
||||
default:
|
||||
return contracts.TypedExtractionResult[dnd.SpellList]{}, fmt.Errorf("unexpected assembled chunk index %d", req.Chunk.Index)
|
||||
|
||||
@@ -24,6 +24,7 @@ func newProductionComponents() (productionComponents, error) {
|
||||
Inputs: pipeline.NewInputAdapterRegistry(),
|
||||
Chunkers: pipeline.NewChunkerRegistry(),
|
||||
ArtifactCodecs: pipeline.NewArtifactCodecRegistry(),
|
||||
ArtifactEvidence: pipeline.NewArtifactEvidenceRegistry(),
|
||||
Extractors: pipeline.NewExtractorRegistry(),
|
||||
Mergers: pipeline.NewMergerRegistry(),
|
||||
Normalizers: pipeline.NewNormalizerRegistry(),
|
||||
@@ -91,6 +92,7 @@ func catalogFromRegistries(registries pipeline.Registries) pipeline.ModuleCatalo
|
||||
Inputs: registries.Inputs,
|
||||
Chunkers: registries.Chunkers,
|
||||
ArtifactCodecs: registries.ArtifactCodecs,
|
||||
ArtifactEvidence: registries.ArtifactEvidence,
|
||||
Extractors: registries.Extractors,
|
||||
Mergers: registries.Mergers,
|
||||
Normalizers: registries.Normalizers,
|
||||
@@ -105,6 +107,7 @@ func registriesFromCatalog(catalog pipeline.ModuleCatalog) pipeline.Registries {
|
||||
Inputs: catalog.Inputs,
|
||||
Chunkers: catalog.Chunkers,
|
||||
ArtifactCodecs: catalog.ArtifactCodecs,
|
||||
ArtifactEvidence: catalog.ArtifactEvidence,
|
||||
Extractors: catalog.Extractors,
|
||||
Mergers: catalog.Mergers,
|
||||
Normalizers: catalog.Normalizers,
|
||||
@@ -118,6 +121,7 @@ func isEmptyCatalog(catalog pipeline.ModuleCatalog) bool {
|
||||
return catalog.Inputs == nil &&
|
||||
catalog.Chunkers == nil &&
|
||||
catalog.ArtifactCodecs == nil &&
|
||||
catalog.ArtifactEvidence == nil &&
|
||||
catalog.Extractors == nil &&
|
||||
catalog.Mergers == nil &&
|
||||
catalog.Normalizers == nil &&
|
||||
@@ -130,6 +134,7 @@ func isEmptyRegistries(registries pipeline.Registries) bool {
|
||||
return registries.Inputs == nil &&
|
||||
registries.Chunkers == nil &&
|
||||
registries.ArtifactCodecs == nil &&
|
||||
registries.ArtifactEvidence == nil &&
|
||||
registries.Extractors == nil &&
|
||||
registries.Mergers == nil &&
|
||||
registries.Normalizers == nil &&
|
||||
@@ -138,24 +143,13 @@ func isEmptyRegistries(registries pipeline.Registries) bool {
|
||||
registries.Outputs == nil
|
||||
}
|
||||
|
||||
func productionLLMClientFactory(ctx context.Context, cfg config.Config, profileID string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
assets, err := productionPromptAssets()
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
return buildProductionLLMClient(ctx, cfg, profileID, assets)
|
||||
}
|
||||
|
||||
func productionLLMClientFactoryWithAssets(assets *llm.AssetRegistry) LLMClientFactory {
|
||||
return func(ctx context.Context, cfg config.Config, profileID string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
return buildProductionLLMClient(ctx, cfg, profileID, assets)
|
||||
return func(ctx context.Context, cfg config.Config, profileID string, overrides LLMRuntimeOverrides) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
return buildProductionLLMClient(ctx, cfg, profileID, overrides, assets)
|
||||
}
|
||||
}
|
||||
|
||||
func buildProductionLLMClient(ctx context.Context, cfg config.Config, profileID string, assets *llm.AssetRegistry) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
func buildProductionLLMClient(ctx context.Context, cfg config.Config, profileID string, overrides LLMRuntimeOverrides, assets *llm.AssetRegistry) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
@@ -163,14 +157,16 @@ func buildProductionLLMClient(ctx context.Context, cfg config.Config, profileID
|
||||
return nil, nil, fmt.Errorf("production asset registry must not be nil")
|
||||
}
|
||||
recorder := llm.NewLLMProfileRecorder()
|
||||
client, err := llm.NewScriptoriumClient(llm.ScriptoriumClientConfig{
|
||||
ProfileDir: cfg.Scriptorium.ProfileDir,
|
||||
ProfileFile: cfg.Scriptorium.ProfileFile,
|
||||
client, err := llm.NewPromptKitClient(llm.PromptKitClientConfig{
|
||||
ProfileDir: cfg.PromptKit.ProfileDir,
|
||||
ProfileFile: cfg.PromptKit.ProfileFile,
|
||||
LocalBackend: mapPromptKitLocalBackend(cfg.PromptKit.LocalBackend),
|
||||
Assets: assets,
|
||||
Recorder: recorder,
|
||||
ReasoningEffort: overrides.ReasoningEffort,
|
||||
})
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("create Scriptorium-backed LLM client: %w", err)
|
||||
return nil, nil, fmt.Errorf("create PromptKit-backed LLM client: %w", err)
|
||||
}
|
||||
scheduler, err := llm.NewScheduler(cfg.Concurrency.TotalLLM)
|
||||
if err != nil {
|
||||
|
||||
@@ -154,6 +154,23 @@ func TestConfigValidateResolvesPipelineAndChecksSelection(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestConfigValidatePipelineDefaultProfileIsOffline(t *testing.T) {
|
||||
configPath := writeCommandConfigContent(t, `version: 4
|
||||
pipelines:
|
||||
demo:
|
||||
llm_profile: dnd-extraction
|
||||
input: seriatim
|
||||
artifacts:
|
||||
spells:
|
||||
extract: dnd/spells
|
||||
`)
|
||||
var stdout, stderr bytes.Buffer
|
||||
code := RunWithOptions([]string{"config", "validate", "--config", configPath, "--pipeline", "demo"}, &stdout, &stderr, Options{})
|
||||
if code != 0 || !strings.Contains(stdout.String(), "valid for pipeline \"demo\"") || stderr.Len() != 0 {
|
||||
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestPipelinesListSortsNormalizedIDsInTextAndJSON(t *testing.T) {
|
||||
configPath := writeCommandConfig(t, " zeta ", "alpha")
|
||||
options := commandContractOptions(t)
|
||||
@@ -214,13 +231,13 @@ func commandContractOptionsWithLookup(t *testing.T, lookup func(string) (string,
|
||||
|
||||
func writeCommandConfig(t *testing.T, firstID, secondID string) string {
|
||||
t.Helper()
|
||||
content := fmt.Sprintf("version: 3\npipelines:\n %q:\n input: seriatim\n %q:\n input: seriatim\n", firstID, secondID)
|
||||
content := fmt.Sprintf("version: 4\npipelines:\n %q:\n input: seriatim\n %q:\n input: seriatim\n", firstID, secondID)
|
||||
return writeCommandConfigContent(t, content)
|
||||
}
|
||||
|
||||
func writeResolvableCommandConfig(t *testing.T) string {
|
||||
t.Helper()
|
||||
return writeCommandConfigContent(t, `version: 3
|
||||
return writeCommandConfigContent(t, `version: 4
|
||||
pipelines:
|
||||
demo:
|
||||
input: seriatim
|
||||
|
||||
@@ -15,8 +15,7 @@ import (
|
||||
|
||||
func TestProductionCombatConfigurationResolvesTypedLane(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
configPath := repositoryPath("examples", "dnd-combat-turns.config.yml")
|
||||
cfg := loadMaintainedExample(t, configPath)
|
||||
cfg := productionCombatContractConfig()
|
||||
effective, err := cfg.Resolve(config.ResolveInput{PipelineID: "dnd-combat", Catalog: catalogFromRegistries(components.registries)})
|
||||
if err != nil {
|
||||
t.Fatalf("Resolve() error = %v, want nil", err)
|
||||
@@ -49,23 +48,27 @@ func TestProductionCombatConfigurationResolvesTypedLane(t *testing.T) {
|
||||
if !ok || codecSpec.Schema.ID != "notarius.dnd.combat_turns" || codecSpec.Schema.Version != "v1" {
|
||||
t.Fatalf("combat codec spec = %#v, want compatible durable schema", codecSpec)
|
||||
}
|
||||
if !hasReferenceSlot(extractSpec.ReferenceSlots, "npcs") || !hasReferenceSlot(normalizeSpec.ReferenceSlots, "npcs") {
|
||||
t.Fatalf("combat reference slots = %#v / %#v, want stage-local NPC slots", extractSpec.ReferenceSlots, normalizeSpec.ReferenceSlots)
|
||||
if !hasReferenceSlot(extractSpec.ReferenceSlots, "npcs") || !hasReferenceSlot(extractSpec.ReferenceSlots, "scene_descriptions") || !hasReferenceSlot(normalizeSpec.ReferenceSlots, "npcs") {
|
||||
t.Fatalf("combat reference slots = %#v / %#v, want extraction scene and NPC slots plus normalization NPC slot", extractSpec.ReferenceSlots, normalizeSpec.ReferenceSlots)
|
||||
}
|
||||
sceneSlot := referenceSlot(extractSpec.ReferenceSlots, "scene_descriptions")
|
||||
if !sceneSlot.Required || !reflect.DeepEqual(sceneSlot.AcceptedMediaTypes, []string{"application/json"}) || !reflect.DeepEqual(sceneSlot.AcceptedArtifactKinds, []contracts.ArtifactKind{dnd.SceneDescriptionListKind}) || sceneSlot.MaxBytes != 1048576 {
|
||||
t.Fatalf("scene description slot = %#v, want required approved scene artifact", sceneSlot)
|
||||
}
|
||||
|
||||
wantExtractChain := []pipeline.ModuleBinding{
|
||||
pipeline.Binding("generic/valid_json"),
|
||||
pipeline.Binding("generic/valid_json_schema"),
|
||||
pipeline.Binding("extract/dnd/combat-turns/shape"),
|
||||
pipeline.Binding("extract/dnd/combat-turns/source_refs"),
|
||||
pipeline.Binding("generic/valid_json_schema"),
|
||||
pipeline.Binding("extract/dnd/combat-turns/source_relatedness"),
|
||||
}
|
||||
wantNormalizeChain := []pipeline.ModuleBinding{
|
||||
pipeline.Binding("generic/valid_json"),
|
||||
pipeline.Binding("generic/valid_json_schema"),
|
||||
pipeline.Binding("extract/dnd/combat-turns/shape"),
|
||||
pipeline.Binding("normalize/dnd/combat-turns/invariants"),
|
||||
pipeline.Binding("extract/dnd/combat-turns/source_refs"),
|
||||
pipeline.Binding("generic/valid_json_schema"),
|
||||
pipeline.Binding("extract/dnd/combat-turns/source_relatedness"),
|
||||
}
|
||||
if got := validatorChain(effective.ResolvedPipeline, pipeline.StageExtract, combatextract.Key); !reflect.DeepEqual(got, wantExtractChain) {
|
||||
@@ -90,16 +93,26 @@ func TestProductionCombatConfigurationResolvesTypedLane(t *testing.T) {
|
||||
t.Fatalf("Resolve(bound references) error = %v, want nil", err)
|
||||
}
|
||||
boundLane := bound.ResolvedPipeline.Steps[0].ArtifactLanes[0]
|
||||
if len(boundLane.ExtractReferences.Bindings) != 1 || len(boundLane.NormalizeReferences.Bindings) != 1 || boundLane.ExtractReferences.Bindings[0].SlotName != "npcs" || boundLane.NormalizeReferences.Bindings[0].SlotName != "npcs" {
|
||||
t.Fatalf("bound combat references = %#v / %#v, want one independent NPC binding per stage", boundLane.ExtractReferences, boundLane.NormalizeReferences)
|
||||
if len(boundLane.ExtractReferences.Bindings) != 2 || len(boundLane.NormalizeReferences.Bindings) != 1 || !hasReferenceBinding(boundLane.ExtractReferences.Bindings, "npcs") || !hasReferenceBinding(boundLane.ExtractReferences.Bindings, "scene_descriptions") || !hasReferenceBinding(boundLane.NormalizeReferences.Bindings, "npcs") {
|
||||
t.Fatalf("bound combat references = %#v / %#v, want extraction scene and NPC bindings plus normalization NPC binding", boundLane.ExtractReferences, boundLane.NormalizeReferences)
|
||||
}
|
||||
}
|
||||
|
||||
func TestProductionCombatConfigurationRequiresSceneDescriptions(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
cfg := productionCombatContractConfig()
|
||||
profile := cfg.Pipelines["dnd-combat"]
|
||||
profile.References = nil
|
||||
cfg.Pipelines["dnd-combat"] = profile
|
||||
if _, err := cfg.Resolve(config.ResolveInput{PipelineID: "dnd-combat", Catalog: catalogFromRegistries(components.registries)}); err == nil || !strings.Contains(err.Error(), "scene_descriptions") || !strings.Contains(err.Error(), "required") {
|
||||
t.Fatalf("Resolve() error = %v, want required scene reference failure", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestProductionCombatConfigurationRejectsLooseOptionsAndLaneValidators(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
configPath := repositoryPath("examples", "dnd-combat-turns.config.yml")
|
||||
resolve := func(mutate func(*pipeline.PipelineProfile)) error {
|
||||
cfg := loadMaintainedExample(t, configPath)
|
||||
cfg := productionCombatContractConfig()
|
||||
profile := cfg.Pipelines["dnd-combat"]
|
||||
mutate(&profile)
|
||||
cfg.Pipelines["dnd-combat"] = profile
|
||||
@@ -131,7 +144,7 @@ func TestProductionCombatConfigurationRejectsLooseOptionsAndLaneValidators(t *te
|
||||
|
||||
func TestProductionCombatConfigurationResolvesTypedUnconditionalValidators(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
cfg := loadMaintainedExample(t, repositoryPath("examples", "dnd-combat-turns.config.yml"))
|
||||
cfg := productionCombatContractConfig()
|
||||
profile := cfg.Pipelines["dnd-combat"]
|
||||
lane := profile.Artifacts["combat"]
|
||||
lane.Extract.Validators = pipeline.ValidatorOverride{Set: true, Validators: []pipeline.ModuleBinding{pipeline.Binding("generic/always_accept")}}
|
||||
@@ -150,6 +163,23 @@ func TestProductionCombatConfigurationResolvesTypedUnconditionalValidators(t *te
|
||||
}
|
||||
}
|
||||
|
||||
func productionCombatContractConfig() config.Config {
|
||||
cfg := config.Default()
|
||||
cfg.Pipelines["dnd-combat"] = pipeline.PipelineProfile{
|
||||
ID: "dnd-combat",
|
||||
Input: pipeline.Binding("seriatim"),
|
||||
Chunk: pipeline.Binding(pipeline.DefaultChunkModule),
|
||||
References: map[string]pipeline.ReferenceSource{"scene_descriptions": pipeline.ExternalReference("scenes.json")},
|
||||
Artifacts: map[string]pipeline.ArtifactLaneProfile{
|
||||
"combat": {
|
||||
Extract: pipeline.ModuleBinding{Module: combatextract.Key, Retries: 2},
|
||||
Normalize: pipeline.Binding(combatnormalize.Key),
|
||||
},
|
||||
},
|
||||
}
|
||||
return cfg
|
||||
}
|
||||
|
||||
func hasReferenceSlot(slots []contracts.ReferenceSlot, name string) bool {
|
||||
for _, slot := range slots {
|
||||
if slot.Name == name {
|
||||
@@ -158,3 +188,21 @@ func hasReferenceSlot(slots []contracts.ReferenceSlot, name string) bool {
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func hasReferenceBinding(bindings []pipeline.ReferenceBinding, name string) bool {
|
||||
for _, binding := range bindings {
|
||||
if binding.SlotName == name {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func referenceSlot(slots []contracts.ReferenceSlot, name string) contracts.ReferenceSlot {
|
||||
for _, slot := range slots {
|
||||
if slot.Name == name {
|
||||
return slot
|
||||
}
|
||||
}
|
||||
return contracts.ReferenceSlot{}
|
||||
}
|
||||
|
||||
135
internal/cli/dnd_interactions_contract_test.go
Normal file
135
internal/cli/dnd_interactions_contract_test.go
Normal file
@@ -0,0 +1,135 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
interactioncodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/npcinteractions"
|
||||
interactionextract "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/npcinteractions"
|
||||
npcextract "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/npcs"
|
||||
interactionnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/npcinteractions"
|
||||
npcnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/npcs"
|
||||
)
|
||||
|
||||
func TestProductionNPCInteractionPipelineResolvesAndPrepares(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
resolved, err := pipeline.ResolvePipeline(npcInteractionProfile(pipeline.GeneratedReference("npcs", "npcs")), pipeline.ResolveOptions{}, catalogFromRegistries(components.registries))
|
||||
if err != nil {
|
||||
t.Fatalf("ResolvePipeline() error = %v", err)
|
||||
}
|
||||
if len(resolved.Steps) != 2 || len(resolved.Steps[1].ArtifactLanes) != 1 {
|
||||
t.Fatalf("resolved pipeline = %#v", resolved)
|
||||
}
|
||||
lane := resolved.Steps[1].ArtifactLanes[0]
|
||||
if lane.ArtifactKind != dnd.NPCInteractionListKind || lane.Extract.Module != interactionextract.Key || lane.Normalize.Module != interactionnormalize.Key {
|
||||
t.Fatalf("interaction lane = %#v", lane)
|
||||
}
|
||||
for _, bindings := range [][]pipeline.ReferenceBinding{lane.ExtractReferences.Bindings, lane.NormalizeReferences.Bindings} {
|
||||
if len(bindings) != 1 || bindings[0].SlotName != "npcs" || bindings[0].Artifact == nil || bindings[0].Artifact.Step != "npcs" || bindings[0].Artifact.Lane != "npcs" {
|
||||
t.Fatalf("generated bindings = %#v", bindings)
|
||||
}
|
||||
}
|
||||
if _, err := pipeline.Prepare(resolved, components.registries, pipeline.ModuleDependencies{LLM: &productionFakeLLMClient{}}); err != nil {
|
||||
t.Fatalf("Prepare() error = %v", err)
|
||||
}
|
||||
|
||||
catalog := catalogFromRegistries(components.registries)
|
||||
codecSpec, ok := catalog.ArtifactCodecs.Spec(dnd.NPCInteractionListKind)
|
||||
if !ok || codecSpec.Schema.ID != interactioncodec.SchemaID || codecSpec.Schema.Version != interactioncodec.SchemaVersion {
|
||||
t.Fatalf("NPC interaction codec spec = %#v", codecSpec)
|
||||
}
|
||||
}
|
||||
|
||||
func TestProductionNPCInteractionReferencesRequireEarlierCompatibleProducer(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
catalog := catalogFromRegistries(components.registries)
|
||||
laterProfile := npcInteractionProfile(pipeline.GeneratedReference("npcs", "npcs"))
|
||||
laterProfile.Steps[0].ID = "seed"
|
||||
laterProfile.Steps[0].Artifacts["seed"] = laterProfile.Steps[0].Artifacts["npcs"]
|
||||
delete(laterProfile.Steps[0].Artifacts, "npcs")
|
||||
laterProfile.Steps = append(laterProfile.Steps, pipeline.PipelineStepProfile{ID: "future", Artifacts: map[string]pipeline.ArtifactLaneProfile{
|
||||
"npcs": {Extract: pipeline.Binding(npcextract.Key), Normalize: pipeline.Binding(npcnormalize.Key)},
|
||||
}})
|
||||
laterProfile.Steps[1].References["npcs"] = pipeline.GeneratedReference("future", "npcs")
|
||||
tests := []struct {
|
||||
name string
|
||||
profile pipeline.PipelineProfile
|
||||
want string
|
||||
}{
|
||||
{name: "missing", profile: npcInteractionProfile(pipeline.ReferenceSource{}), want: "source must not be empty"},
|
||||
{name: "same step", profile: npcInteractionProfile(pipeline.GeneratedReference("interactions", "interactions")), want: "earlier step"},
|
||||
{name: "later step", profile: laterProfile, want: "earlier step"},
|
||||
{name: "wrong artifact kind", profile: npcInteractionProfile(pipeline.GeneratedReference("npcs", "npcs")), want: "does not accept artifact kind"},
|
||||
}
|
||||
tests[3].profile.Steps[0].Artifacts["npcs"] = pipeline.ArtifactLaneProfile{Extract: pipeline.Binding("dnd/spells")}
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
_, err := pipeline.ResolvePipeline(test.profile, pipeline.ResolveOptions{}, catalog)
|
||||
if err == nil || !strings.Contains(err.Error(), test.want) {
|
||||
t.Fatalf("ResolvePipeline() error = %v, want %q", err, test.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestProductionNPCInteractionReferencesRejectIncompatibleExternalRegistries(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
catalog := catalogFromRegistries(components.registries)
|
||||
root := t.TempDir()
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
file string
|
||||
content string
|
||||
prepare bool
|
||||
want string
|
||||
}{
|
||||
{name: "media type", file: "registry.txt", content: "not JSON", want: "media type"},
|
||||
{name: "artifact schema", file: "registry.json", content: `{"npcs":[{"name":"missing required fields"}]}`, prepare: true, want: "NPC registry"},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
path := filepath.Join(root, test.file)
|
||||
if err := os.WriteFile(path, []byte(test.content), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
resolved, err := pipeline.ResolvePipeline(npcInteractionProfile(pipeline.ExternalReference(path)), pipeline.ResolveOptions{}, catalog)
|
||||
if err != nil {
|
||||
t.Fatalf("ResolvePipeline() error = %v", err)
|
||||
}
|
||||
materialized, _, err := pipeline.MaterializeReferences(resolved, catalog, pipeline.ReferenceMaterializationOptions{})
|
||||
if !test.prepare {
|
||||
if err == nil || !strings.Contains(err.Error(), test.want) {
|
||||
t.Fatalf("MaterializeReferences() error = %v, want %q", err, test.want)
|
||||
}
|
||||
return
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatalf("MaterializeReferences() error = %v", err)
|
||||
}
|
||||
if _, err := pipeline.Prepare(materialized, components.registries, pipeline.ModuleDependencies{LLM: &productionFakeLLMClient{}}); err == nil || !strings.Contains(err.Error(), test.want) {
|
||||
t.Fatalf("Prepare() error = %v, want %q", err, test.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func npcInteractionProfile(reference pipeline.ReferenceSource) pipeline.PipelineProfile {
|
||||
profile := pipeline.PipelineProfile{
|
||||
ID: "dnd-npc-interactions",
|
||||
Input: pipeline.Binding("seriatim"),
|
||||
Chunk: pipeline.ModuleBinding{Module: "generic", Options: map[string]any{"max_units": 1}},
|
||||
Output: pipeline.Binding("json"),
|
||||
Steps: []pipeline.PipelineStepProfile{
|
||||
{ID: "npcs", Artifacts: map[string]pipeline.ArtifactLaneProfile{
|
||||
"npcs": {Extract: pipeline.Binding(npcextract.Key), Normalize: pipeline.Binding(npcnormalize.Key)},
|
||||
}},
|
||||
{ID: "interactions", References: map[string]pipeline.ReferenceSource{"npcs": reference}, Artifacts: map[string]pipeline.ArtifactLaneProfile{
|
||||
"interactions": {Extract: pipeline.Binding(interactionextract.Key), Normalize: pipeline.Binding(interactionnormalize.Key)},
|
||||
}},
|
||||
},
|
||||
}
|
||||
return profile
|
||||
}
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
npccodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/npcs"
|
||||
npcextract "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/npcs"
|
||||
npcnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/npcs"
|
||||
)
|
||||
@@ -15,8 +16,7 @@ import (
|
||||
func TestProductionNPCConfigurationResolvesTypedLane(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
catalog := catalogFromRegistries(components.registries)
|
||||
configPath := repositoryPath("examples", "dnd-npcs.config.yml")
|
||||
cfg := loadMaintainedExample(t, configPath)
|
||||
cfg := productionNPCContractConfig()
|
||||
effective, err := cfg.Resolve(config.ResolveInput{PipelineID: "dnd-session", Catalog: catalog})
|
||||
if err != nil {
|
||||
t.Fatalf("Resolve() error = %v, want nil", err)
|
||||
@@ -47,20 +47,24 @@ func TestProductionNPCConfigurationResolvesTypedLane(t *testing.T) {
|
||||
if !ok || !reflect.DeepEqual(normalizeSpec.Requires, []string{"merged"}) || !reflect.DeepEqual(normalizeSpec.Provides, []string{"normalized"}) {
|
||||
t.Fatalf("NPC normalizer spec = %#v, want merged/normalized capabilities", normalizeSpec)
|
||||
}
|
||||
codecSpec, ok := catalog.ArtifactCodecs.Spec(dnd.NPCListKind)
|
||||
if !ok || codecSpec.Kind != dnd.NPCListKind || codecSpec.Schema.ID != npccodec.SchemaID || codecSpec.Schema.Version != npccodec.SchemaVersion {
|
||||
t.Fatalf("NPC codec spec = %#v, want typed v1 durable schema", codecSpec)
|
||||
}
|
||||
|
||||
wantExtractChain := []pipeline.ModuleBinding{
|
||||
pipeline.Binding("generic/valid_json"),
|
||||
pipeline.Binding("generic/valid_json_schema"),
|
||||
pipeline.Binding("extract/dnd/npcs/shape"),
|
||||
pipeline.Binding("extract/dnd/npcs/source_refs"),
|
||||
pipeline.Binding("generic/valid_json_schema"),
|
||||
pipeline.Binding("extract/dnd/npcs/source_relatedness"),
|
||||
}
|
||||
wantNormalizeChain := []pipeline.ModuleBinding{
|
||||
pipeline.Binding("generic/valid_json"),
|
||||
pipeline.Binding("generic/valid_json_schema"),
|
||||
pipeline.Binding("extract/dnd/npcs/shape"),
|
||||
pipeline.Binding("normalize/dnd/npcs/identity"),
|
||||
pipeline.Binding("extract/dnd/npcs/source_refs"),
|
||||
pipeline.Binding("generic/valid_json_schema"),
|
||||
pipeline.Binding("extract/dnd/npcs/source_relatedness"),
|
||||
}
|
||||
if got := validatorChain(effective.ResolvedPipeline, pipeline.StageExtract, npcextract.Key); !reflect.DeepEqual(got, wantExtractChain) {
|
||||
@@ -76,9 +80,8 @@ func TestProductionNPCConfigurationResolvesTypedLane(t *testing.T) {
|
||||
|
||||
func TestProductionNPCConfigurationValidatesOptionsReferencesAndPlacement(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
configPath := repositoryPath("examples", "dnd-npcs.config.yml")
|
||||
resolve := func(mutate func(*pipeline.PipelineProfile)) error {
|
||||
cfg := loadMaintainedExample(t, configPath)
|
||||
cfg := productionNPCContractConfig()
|
||||
profile := cfg.Pipelines["dnd-session"]
|
||||
mutate(&profile)
|
||||
cfg.Pipelines["dnd-session"] = profile
|
||||
@@ -118,6 +121,22 @@ func TestProductionNPCConfigurationValidatesOptionsReferencesAndPlacement(t *tes
|
||||
}
|
||||
}
|
||||
|
||||
func productionNPCContractConfig() config.Config {
|
||||
cfg := config.Default()
|
||||
cfg.Pipelines["dnd-session"] = pipeline.PipelineProfile{
|
||||
ID: "dnd-session",
|
||||
Input: pipeline.Binding("seriatim"),
|
||||
Chunk: pipeline.Binding(pipeline.DefaultChunkModule),
|
||||
Artifacts: map[string]pipeline.ArtifactLaneProfile{
|
||||
"npcs": {
|
||||
Extract: pipeline.ModuleBinding{Module: npcextract.Key, Retries: 2},
|
||||
Normalize: pipeline.Binding(npcnormalize.Key),
|
||||
},
|
||||
},
|
||||
}
|
||||
return cfg
|
||||
}
|
||||
|
||||
func validatorChain(resolved pipeline.ResolvedPipeline, stage pipeline.ModuleStage, module string) []pipeline.ModuleBinding {
|
||||
for _, chain := range resolved.ValidatorChains {
|
||||
if chain.Stage == stage && chain.ModuleKey == module {
|
||||
|
||||
147
internal/cli/dnd_scene_descriptions_contract_test.go
Normal file
147
internal/cli/dnd_scene_descriptions_contract_test.go
Normal file
@@ -0,0 +1,147 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"reflect"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
scenecodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/scenedescriptions"
|
||||
sceneextract "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/scenedescriptions"
|
||||
scenenormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/scenedescriptions"
|
||||
)
|
||||
|
||||
func TestProductionSceneDescriptionWorkflow(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
cfg := config.Default()
|
||||
cfg.Pipelines["scene-descriptions"] = pipeline.PipelineProfile{
|
||||
ID: "scene-descriptions",
|
||||
Input: pipeline.Binding("seriatim"),
|
||||
Chunk: pipeline.ModuleBinding{Module: "generic", Options: map[string]any{"max_units": 1}},
|
||||
Output: pipeline.Binding("json"),
|
||||
Artifacts: map[string]pipeline.ArtifactLaneProfile{
|
||||
"scene-descriptions": {
|
||||
Extract: pipeline.ModuleBinding{Module: sceneextract.Key, LLMProfile: "scene-description-profile"},
|
||||
Normalize: pipeline.Binding(scenenormalize.Key),
|
||||
},
|
||||
},
|
||||
}
|
||||
effective, err := cfg.Resolve(config.ResolveInput{PipelineID: "scene-descriptions", Catalog: catalogFromRegistries(components.registries)})
|
||||
if err != nil {
|
||||
t.Fatalf("Resolve() error = %v", err)
|
||||
}
|
||||
lane := effective.ResolvedPipeline.Steps[0].ArtifactLanes[0]
|
||||
if lane.ArtifactKind != dnd.SceneDescriptionListKind || lane.Extract.Module != sceneextract.Key || lane.Merge.Module != pipeline.DefaultMergeModule || lane.Normalize.Module != scenenormalize.Key {
|
||||
t.Fatalf("resolved lane = %#v, want production scene-description composition", lane)
|
||||
}
|
||||
if len(lane.ExtractReferences.Bindings) != 0 || len(lane.NormalizeReferences.Bindings) != 0 {
|
||||
t.Fatalf("resolved references = %#v / %#v, want no generated or required references", lane.ExtractReferences, lane.NormalizeReferences)
|
||||
}
|
||||
|
||||
llmClient := &sceneDescriptionLLM{}
|
||||
prepared, err := pipeline.Prepare(effective.ResolvedPipeline, components.registries, pipeline.ModuleDependencies{LLM: llmClient})
|
||||
if err != nil {
|
||||
t.Fatalf("Prepare() error = %v", err)
|
||||
}
|
||||
output, err := pipeline.New().Run(context.Background(), pipeline.RunInput{
|
||||
Prepared: prepared,
|
||||
RawInput: readRepositoryFile(t, "examples", "seriatim-minimal-transcript.json"),
|
||||
ChunkCacheMode: pipeline.ChunkCacheBypass,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("Run() error = %v", err)
|
||||
}
|
||||
if output.Manifest.ValidationStatus != "approved" || len(output.Rejected) != 0 || len(output.NormalizeOutputs) != 1 {
|
||||
t.Fatalf("run output = %#v, want one approved normalized artifact", output)
|
||||
}
|
||||
wantProfiles := []artifacts.LLMProfileManifest{{
|
||||
ID: "scene-description-profile",
|
||||
Provider: "promptkit",
|
||||
Model: "deterministic",
|
||||
}}
|
||||
if !reflect.DeepEqual(output.Manifest.LLMProfiles, wantProfiles) {
|
||||
t.Fatalf("manifest LLM profiles = %#v, want %#v", output.Manifest.LLMProfiles, wantProfiles)
|
||||
}
|
||||
normalizedOutput := output.NormalizeOutputs[0]
|
||||
if normalizedOutput.NormalizerKey != scenenormalize.Key || normalizedOutput.Artifact.Kind != dnd.SceneDescriptionListKind || normalizedOutput.Artifact.Schema.ID != scenecodec.SchemaID || normalizedOutput.Artifact.Schema.Name != scenecodec.SchemaName || normalizedOutput.Artifact.Schema.Version != scenecodec.SchemaVersion {
|
||||
t.Fatalf("normalized output = %#v, want registered durable scene-description schema", normalizedOutput)
|
||||
}
|
||||
|
||||
var value dnd.SceneDescriptionList
|
||||
if err := json.Unmarshal(normalizedOutput.Artifact.Content, &value); err != nil {
|
||||
t.Fatalf("decode normalized artifact: %v", err)
|
||||
}
|
||||
want := dnd.SceneDescriptionList{Scenes: []dnd.SceneDescription{
|
||||
{ID: "chunk-000001", SourceRef: source.SourceRef{SourceID: "session-alpha", StartUnitID: 1, EndUnitID: 1}, Kind: dnd.SceneKindNarrative, Title: "Aria casts Cure Wounds", Summary: "Aria casts Cure Wounds."},
|
||||
{ID: "chunk-000002", SourceRef: source.SourceRef{SourceID: "session-alpha", StartUnitID: 2, EndUnitID: 2}, Kind: dnd.SceneKindCombat, Title: "Bandit mage casts Shield", Summary: "The bandit mage casts Shield."},
|
||||
}}
|
||||
if !reflect.DeepEqual(value, want) {
|
||||
t.Fatalf("normalized scene descriptions = %#v, want %#v", value, want)
|
||||
}
|
||||
durable := decodeAssembledOutput[dnd.SceneDescriptionList](t, output.OutputFiles, "lanes/scene-descriptions.json")
|
||||
if !reflect.DeepEqual(durable, want) {
|
||||
t.Fatalf("durable output payload = %#v, want %#v", durable, want)
|
||||
}
|
||||
if len(output.Warnings) != 0 {
|
||||
t.Fatalf("warnings = %#v, want grounded descriptions without warnings", output.Warnings)
|
||||
}
|
||||
}
|
||||
|
||||
type sceneDescriptionLLM struct {
|
||||
mu sync.Mutex
|
||||
profile *artifacts.LLMProfileManifest
|
||||
}
|
||||
|
||||
func (client *sceneDescriptionLLM) CompleteStructured(ctx context.Context, req contracts.StructuredCompletionRequest, out any) (contracts.StructuredCompletionResponse, error) {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return contracts.StructuredCompletionResponse{}, err
|
||||
}
|
||||
if req.PromptID != sceneextract.PromptID {
|
||||
return contracts.StructuredCompletionResponse{}, fmt.Errorf("unexpected prompt %q", req.PromptID)
|
||||
}
|
||||
transcript := string(req.Inputs["transcript"].Content)
|
||||
var content string
|
||||
switch {
|
||||
case strings.Contains(transcript, "Cure Wounds"):
|
||||
content = `{"kind":"narrative","title":" Aria casts Cure Wounds ","summary":" Aria casts Cure Wounds. "}`
|
||||
case strings.Contains(transcript, "Shield"):
|
||||
content = `{"kind":"combat","title":"Bandit mage casts Shield","summary":"The bandit mage casts Shield."}`
|
||||
default:
|
||||
return contracts.StructuredCompletionResponse{}, fmt.Errorf("unexpected transcript material %q", transcript)
|
||||
}
|
||||
if err := json.Unmarshal([]byte(content), out); err != nil {
|
||||
return contracts.StructuredCompletionResponse{}, fmt.Errorf("populate structured response: %w", err)
|
||||
}
|
||||
profile := artifacts.LLMProfileManifest{
|
||||
ID: req.ProfileID,
|
||||
Provider: "promptkit",
|
||||
Model: "deterministic",
|
||||
}
|
||||
client.mu.Lock()
|
||||
client.profile = &profile
|
||||
client.mu.Unlock()
|
||||
return contracts.StructuredCompletionResponse{
|
||||
Content: []byte(content),
|
||||
Provider: profile.Provider,
|
||||
Model: profile.Model,
|
||||
ProfileID: profile.ID,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func (client *sceneDescriptionLLM) LLMProfileManifests() []artifacts.LLMProfileManifest {
|
||||
client.mu.Lock()
|
||||
defer client.mu.Unlock()
|
||||
if client.profile == nil {
|
||||
return nil
|
||||
}
|
||||
return []artifacts.LLMProfileManifest{*client.profile}
|
||||
}
|
||||
@@ -1,18 +1,22 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/debugbundle"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
spellnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/spells"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/seriatim/input/transcript"
|
||||
)
|
||||
|
||||
func TestMaintainedExamplesLoadResolveAndList(t *testing.T) {
|
||||
@@ -20,6 +24,20 @@ func TestMaintainedExamplesLoadResolveAndList(t *testing.T) {
|
||||
for _, example := range maintainedExampleFiles(t) {
|
||||
t.Run(example.name, func(t *testing.T) {
|
||||
cfg := loadMaintainedExample(t, example.path)
|
||||
raw, err := os.ReadFile(example.transcriptPath)
|
||||
if err != nil {
|
||||
t.Fatalf("read maintained transcript %q: %v", example.transcriptPath, err)
|
||||
}
|
||||
document, err := transcript.New().Parse(context.Background(), contracts.ParseRequest{
|
||||
Path: example.transcriptPath,
|
||||
Raw: raw,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("parse maintained transcript %q: %v", example.transcriptPath, err)
|
||||
}
|
||||
if len(document.Units) == 0 {
|
||||
t.Fatalf("maintained transcript %q has no parsed units", example.transcriptPath)
|
||||
}
|
||||
for _, pipelineID := range example.pipelineIDs {
|
||||
effective, err := cfg.Resolve(resolveInputForMaintainedExample(components, pipelineID))
|
||||
if err != nil {
|
||||
@@ -32,11 +50,23 @@ func TestMaintainedExamplesLoadResolveAndList(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("materialize maintained example references for %q: %v", pipelineID, err)
|
||||
}
|
||||
if example.name == "production" {
|
||||
if len(materialized.Steps[0].ArtifactLanes) != 1 ||
|
||||
len(materialized.Steps[0].ArtifactLanes[0].ExtractReferences.ReferenceSet.Slots["spell_catalog"].Items) != 1 ||
|
||||
len(materialized.Steps[0].ArtifactLanes[0].NormalizeReferences.ReferenceSet.Slots["spell_catalog"].Items) != 1 {
|
||||
t.Fatalf("production spell catalog reference was not materialized: %#v", materialized.Steps[0].ArtifactLanes)
|
||||
if example.name == "complete" {
|
||||
if got := exampleStepLaneIDs(materialized); strings.Join(got, "|") != "describe-session:item-events,npcs,scene-descriptions|extract-events:combat-turns,npc-interactions,spells" {
|
||||
t.Fatalf("complete example steps and lanes = %v, want every D&D extractor in the documented two-step composition", got)
|
||||
}
|
||||
spellLane := referenceContractLane(t, materialized, "spells")
|
||||
if len(spellLane.ExtractReferences.ReferenceSet.Slots["spell_catalog"].Items) != 1 ||
|
||||
len(spellLane.NormalizeReferences.ReferenceSet.Slots["spell_catalog"].Items) != 1 {
|
||||
t.Fatalf("complete example spell catalog reference was not materialized: %#v", spellLane)
|
||||
}
|
||||
itemEventLane := referenceContractLane(t, materialized, "item-events")
|
||||
for _, references := range []pipeline.ResolvedReferenceTarget{itemEventLane.ExtractReferences, itemEventLane.NormalizeReferences} {
|
||||
if _, found := references.ReferenceSet.Slots["npcs"]; found {
|
||||
t.Fatalf("item event lane unexpectedly depends on generated NPCs: %#v", itemEventLane)
|
||||
}
|
||||
if _, found := references.ReferenceSet.Slots["scene_descriptions"]; found {
|
||||
t.Fatalf("item event lane unexpectedly depends on generated scene descriptions: %#v", itemEventLane)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -49,6 +79,67 @@ func TestMaintainedExamplesLoadResolveAndList(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestMaintainedConfigurationExampleSet(t *testing.T) {
|
||||
entries, err := os.ReadDir(repositoryPath("examples"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var names []string
|
||||
for _, entry := range entries {
|
||||
if !entry.IsDir() && strings.HasSuffix(entry.Name(), ".config.yml") {
|
||||
names = append(names, entry.Name())
|
||||
}
|
||||
}
|
||||
sort.Strings(names)
|
||||
if got := strings.Join(names, ","); got != "dnd-complete.config.yml,dnd-minimal.config.yml" {
|
||||
t.Fatalf("maintained configuration examples = %q, want only the minimal and complete D&D examples", got)
|
||||
}
|
||||
|
||||
profileEntries, err := os.ReadDir(repositoryPath("examples", "profiles"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
names = names[:0]
|
||||
for _, entry := range profileEntries {
|
||||
if !entry.IsDir() && strings.HasSuffix(entry.Name(), ".yml") {
|
||||
names = append(names, entry.Name())
|
||||
}
|
||||
}
|
||||
sort.Strings(names)
|
||||
if got := strings.Join(names, ","); got != "dnd-extraction.yml" {
|
||||
t.Fatalf("maintained operator profiles = %q, want dnd-extraction.yml", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMaintainedExamplesValidateEffectiveProfilesOffline(t *testing.T) {
|
||||
t.Chdir(repositoryPath())
|
||||
t.Setenv("OPENROUTER_API_KEY", "")
|
||||
for _, example := range maintainedExampleFiles(t) {
|
||||
t.Run(example.name, func(t *testing.T) {
|
||||
var stdout, stderr strings.Builder
|
||||
code := RunWithOptions([]string{
|
||||
"config", "validate", "--config", example.path, "--pipeline", "dnd-session",
|
||||
}, &stdout, &stderr, Options{})
|
||||
if code != 0 || stderr.Len() != 0 || !strings.Contains(stdout.String(), `valid for pipeline "dnd-session"`) {
|
||||
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func exampleStepLaneIDs(resolved pipeline.ResolvedPipeline) []string {
|
||||
result := make([]string, 0, len(resolved.Steps))
|
||||
for _, step := range resolved.Steps {
|
||||
laneIDs := make([]string, 0, len(step.ArtifactLanes))
|
||||
for _, lane := range step.ArtifactLanes {
|
||||
laneIDs = append(laneIDs, lane.ID)
|
||||
}
|
||||
sort.Strings(laneIDs)
|
||||
result = append(result, step.ID+":"+strings.Join(laneIDs, ","))
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
func TestMaintainedMinimalInvocationProducesJSONBundle(t *testing.T) {
|
||||
outputRoot := filepath.Join(t.TempDir(), "output")
|
||||
fake := &productionFakeLLMClient{}
|
||||
@@ -56,7 +147,7 @@ func TestMaintainedMinimalInvocationProducesJSONBundle(t *testing.T) {
|
||||
var stdout, stderr strings.Builder
|
||||
code := RunWithOptions([]string{
|
||||
"run", "dnd-session",
|
||||
"--config", repositoryPath("examples", "dnd-spells.config.yml"),
|
||||
"--config", repositoryPath("examples", "dnd-minimal.config.yml"),
|
||||
"--input", repositoryPath("examples", "seriatim-minimal-transcript.json"),
|
||||
"--only", "spells", "--chunk_cache", "bypass", "--output-dir", outputRoot,
|
||||
}, &stdout, &stderr, options)
|
||||
@@ -130,7 +221,7 @@ func TestMaintainedMalformedInputOnlyRecordsDebugFailureWhenRequested(t *testing
|
||||
options := productionRunOptions(t, &productionFakeLLMClient{})
|
||||
args := []string{
|
||||
"run", "dnd-session",
|
||||
"--config", repositoryPath("examples", "dnd-spells.config.yml"),
|
||||
"--config", repositoryPath("examples", "dnd-minimal.config.yml"),
|
||||
"--input", malformed, "--chunk_cache", "bypass", "--output-dir", outputRoot,
|
||||
}
|
||||
if debug {
|
||||
|
||||
@@ -3,6 +3,7 @@ package cli
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io/fs"
|
||||
"os"
|
||||
"path/filepath"
|
||||
@@ -22,9 +23,24 @@ func TestOversizedNPCRegistryFailsBeforeRuntimeAndCheckpointConstruction(t *test
|
||||
t.Fatal(err)
|
||||
}
|
||||
checkpointRoot := filepath.Join(t.TempDir(), "checkpoints")
|
||||
content := string(readRepositoryFile(t, "examples", "dnd-spells.config.yml"))
|
||||
content = replaceRequiredOnce(t, content, " extract: dnd/spells", " extract:\n module: dnd/spells\n references:\n npcs: "+npcPath)
|
||||
content = replaceRequiredOnce(t, content, " enabled: false\n directory: \"\"", " enabled: true\n directory: "+checkpointRoot)
|
||||
content := fmt.Sprintf(`version: 4
|
||||
cache:
|
||||
chunk_plans:
|
||||
mode: bypass
|
||||
checkpoints:
|
||||
enabled: true
|
||||
directory: %q
|
||||
pipelines:
|
||||
dnd-session:
|
||||
input: seriatim
|
||||
artifacts:
|
||||
spells:
|
||||
extract:
|
||||
module: dnd/spells
|
||||
references:
|
||||
npcs: %q
|
||||
normalize: dnd/spells
|
||||
`, checkpointRoot, npcPath)
|
||||
configPath := filepath.Join(t.TempDir(), "config.yml")
|
||||
if err := os.WriteFile(configPath, []byte(content), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
@@ -35,7 +51,7 @@ func TestOversizedNPCRegistryFailsBeforeRuntimeAndCheckpointConstruction(t *test
|
||||
options := Options{
|
||||
Catalog: catalogFromRegistries(components.registries),
|
||||
Registries: components.registries,
|
||||
LLMClientFactory: func(context.Context, config.Config, string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
LLMClientFactory: func(context.Context, config.Config, string, LLMRuntimeOverrides) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
llmConstructed = true
|
||||
return nil, nil, errors.New("LLM client must not be constructed")
|
||||
},
|
||||
|
||||
@@ -6,6 +6,8 @@ import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"io/fs"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"reflect"
|
||||
@@ -13,22 +15,30 @@ import (
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"testing/fstest"
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/chunkmap"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/chunk/scenes"
|
||||
combatcodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/combatturns"
|
||||
itemeventcodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/itemevents"
|
||||
spellcodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/spells"
|
||||
combatextract "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/combatturns"
|
||||
itemeventextract "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/itemevents"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/spells"
|
||||
combatnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/combatturns"
|
||||
itemeventnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/itemevents"
|
||||
spellnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/spells"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/generic/normalize/noop"
|
||||
"gitea.maximumdirect.net/eric/promptkit"
|
||||
)
|
||||
|
||||
func TestProductionCatalogCoversMaintainedConfigurations(t *testing.T) {
|
||||
@@ -37,9 +47,9 @@ func TestProductionCatalogCoversMaintainedConfigurations(t *testing.T) {
|
||||
|
||||
assertProductionContains(t, "inputs", registries.Inputs.RegisteredKeys(), []string{"seriatim"})
|
||||
assertProductionContains(t, "chunkers", registries.Chunkers.RegisteredKeys(), []string{"dnd/scenes", "generic"})
|
||||
assertProductionContains(t, "extractors", registries.Extractors.RegisteredKeys(), []string{"dnd/spells", "dnd/npcs", combatextract.Key})
|
||||
assertProductionContains(t, "extractors", registries.Extractors.RegisteredKeys(), []string{"dnd/spells", "dnd/npcs", combatextract.Key, itemeventextract.Key})
|
||||
assertProductionContains(t, "mergers", registries.Mergers.RegisteredKeys(), []string{"appendorder"})
|
||||
assertProductionContains(t, "normalizers", registries.Normalizers.RegisteredKeys(), []string{"noop", spellnormalize.Key, "dnd/npcs", combatnormalize.Key})
|
||||
assertProductionContains(t, "normalizers", registries.Normalizers.RegisteredKeys(), []string{"noop", spellnormalize.Key, "dnd/npcs", combatnormalize.Key, itemeventnormalize.Key})
|
||||
assertProductionContains(t, "outputs", registries.Outputs.RegisteredKeys(), []string{"json"})
|
||||
assertProductionContains(t, "validators", registries.Validators.RegisteredKeys(), []string{
|
||||
"extract/dnd/spells/catalog",
|
||||
@@ -50,23 +60,28 @@ func TestProductionCatalogCoversMaintainedConfigurations(t *testing.T) {
|
||||
"extract/dnd/combat-turns/source_refs",
|
||||
"extract/dnd/combat-turns/source_relatedness",
|
||||
"normalize/dnd/combat-turns/invariants",
|
||||
"extract/dnd/item-events/shape",
|
||||
"extract/dnd/item-events/source_refs",
|
||||
"extract/dnd/item-events/source_relatedness",
|
||||
"normalize/dnd/item-events/invariants",
|
||||
"generic/always_accept",
|
||||
"generic/always_reject",
|
||||
"generic/valid_json",
|
||||
"generic/valid_json_schema",
|
||||
})
|
||||
assertProductionContains(t, "artifact codec kinds", registries.ArtifactCodecs.RegisteredKinds(), []contracts.ArtifactKind{dnd.SpellListKind, dnd.NPCListKind, dnd.CombatTurnListKind})
|
||||
assertProductionContains(t, "merger variants", registries.Mergers.RegisteredArtifactKinds(pipeline.DefaultMergeModule), []contracts.ArtifactKind{dnd.SpellListKind, dnd.NPCListKind, dnd.CombatTurnListKind})
|
||||
assertProductionContains(t, "normalizer variants", registries.Normalizers.RegisteredArtifactKinds(pipeline.DefaultNormalizeModule), []contracts.ArtifactKind{dnd.SpellListKind, dnd.NPCListKind, dnd.CombatTurnListKind})
|
||||
assertProductionContains(t, "artifact codec kinds", registries.ArtifactCodecs.RegisteredKinds(), []contracts.ArtifactKind{dnd.SpellListKind, dnd.NPCListKind, dnd.CombatTurnListKind, dnd.ItemEventListKind})
|
||||
assertProductionContains(t, "merger variants", registries.Mergers.RegisteredArtifactKinds(pipeline.DefaultMergeModule), []contracts.ArtifactKind{dnd.SpellListKind, dnd.NPCListKind, dnd.CombatTurnListKind, dnd.ItemEventListKind})
|
||||
assertProductionContains(t, "normalizer variants", registries.Normalizers.RegisteredArtifactKinds(pipeline.DefaultNormalizeModule), []contracts.ArtifactKind{dnd.SpellListKind, dnd.NPCListKind, dnd.CombatTurnListKind, dnd.ItemEventListKind})
|
||||
assertProductionContains(t, "spell normalizer variants", registries.Normalizers.RegisteredArtifactKinds(spellnormalize.Key), []contracts.ArtifactKind{dnd.SpellListKind})
|
||||
assertProductionContains(t, "combat normalizer variants", registries.Normalizers.RegisteredArtifactKinds(combatnormalize.Key), []contracts.ArtifactKind{dnd.CombatTurnListKind})
|
||||
assertProductionContains(t, "item event normalizer variants", registries.Normalizers.RegisteredArtifactKinds(itemeventnormalize.Key), []contracts.ArtifactKind{dnd.ItemEventListKind})
|
||||
|
||||
wantChain := []pipeline.ModuleBinding{
|
||||
pipeline.Binding("generic/valid_json"),
|
||||
pipeline.Binding("generic/valid_json_schema"),
|
||||
pipeline.Binding("extract/dnd/spells/shape"),
|
||||
pipeline.Binding("extract/dnd/spells/catalog"),
|
||||
pipeline.Binding("extract/dnd/spells/source_refs"),
|
||||
pipeline.Binding("generic/valid_json_schema"),
|
||||
pipeline.Binding("extract/dnd/spells/source_relatedness"),
|
||||
}
|
||||
if got := registries.ValidatorChains.Validators(pipeline.StageExtract, spells.Key); !reflect.DeepEqual(got, wantChain) {
|
||||
@@ -77,17 +92,17 @@ func TestProductionCatalogCoversMaintainedConfigurations(t *testing.T) {
|
||||
}
|
||||
combatExtractChain := []pipeline.ModuleBinding{
|
||||
pipeline.Binding("generic/valid_json"),
|
||||
pipeline.Binding("generic/valid_json_schema"),
|
||||
pipeline.Binding("extract/dnd/combat-turns/shape"),
|
||||
pipeline.Binding("extract/dnd/combat-turns/source_refs"),
|
||||
pipeline.Binding("generic/valid_json_schema"),
|
||||
pipeline.Binding("extract/dnd/combat-turns/source_relatedness"),
|
||||
}
|
||||
combatNormalizeChain := []pipeline.ModuleBinding{
|
||||
pipeline.Binding("generic/valid_json"),
|
||||
pipeline.Binding("generic/valid_json_schema"),
|
||||
pipeline.Binding("extract/dnd/combat-turns/shape"),
|
||||
pipeline.Binding("normalize/dnd/combat-turns/invariants"),
|
||||
pipeline.Binding("extract/dnd/combat-turns/source_refs"),
|
||||
pipeline.Binding("generic/valid_json_schema"),
|
||||
pipeline.Binding("extract/dnd/combat-turns/source_relatedness"),
|
||||
}
|
||||
if got := registries.ValidatorChains.Validators(pipeline.StageExtract, combatextract.Key); !reflect.DeepEqual(got, combatExtractChain) {
|
||||
@@ -96,6 +111,27 @@ func TestProductionCatalogCoversMaintainedConfigurations(t *testing.T) {
|
||||
if got := registries.ValidatorChains.Validators(pipeline.StageNormalize, combatnormalize.Key); !reflect.DeepEqual(got, combatNormalizeChain) {
|
||||
t.Fatalf("combat normalize validator chain = %#v, want %#v", got, combatNormalizeChain)
|
||||
}
|
||||
itemEventExtractChain := []pipeline.ModuleBinding{
|
||||
pipeline.Binding("generic/valid_json"),
|
||||
pipeline.Binding("extract/dnd/item-events/shape"),
|
||||
pipeline.Binding("extract/dnd/item-events/source_refs"),
|
||||
pipeline.Binding("generic/valid_json_schema"),
|
||||
pipeline.Binding("extract/dnd/item-events/source_relatedness"),
|
||||
}
|
||||
itemEventNormalizeChain := []pipeline.ModuleBinding{
|
||||
pipeline.Binding("generic/valid_json"),
|
||||
pipeline.Binding("extract/dnd/item-events/shape"),
|
||||
pipeline.Binding("normalize/dnd/item-events/invariants"),
|
||||
pipeline.Binding("extract/dnd/item-events/source_refs"),
|
||||
pipeline.Binding("generic/valid_json_schema"),
|
||||
pipeline.Binding("extract/dnd/item-events/source_relatedness"),
|
||||
}
|
||||
if got := registries.ValidatorChains.Validators(pipeline.StageExtract, itemeventextract.Key); !reflect.DeepEqual(got, itemEventExtractChain) {
|
||||
t.Fatalf("item event extract validator chain = %#v, want %#v", got, itemEventExtractChain)
|
||||
}
|
||||
if got := registries.ValidatorChains.Validators(pipeline.StageNormalize, itemeventnormalize.Key); !reflect.DeepEqual(got, itemEventNormalizeChain) {
|
||||
t.Fatalf("item event normalize validator chain = %#v, want %#v", got, itemEventNormalizeChain)
|
||||
}
|
||||
|
||||
assetNames := productionAssetNames(t, components.assets.PromptFS)
|
||||
requiredAssets := []string{
|
||||
@@ -118,13 +154,50 @@ func TestProductionCatalogCoversMaintainedConfigurations(t *testing.T) {
|
||||
"dnd.combat_turns/sharedassets/common-dnd-system.md",
|
||||
"dnd.combat_turns/sharedassets/common-dnd-transcript.md",
|
||||
"dnd.combat_turns/task.md",
|
||||
"dnd.item_events/dnd.item_events.yaml",
|
||||
"dnd.item_events/instructions.md",
|
||||
"dnd.item_events/sharedassets/common-dnd-extraction-evidence.md",
|
||||
"dnd.item_events/sharedassets/common-dnd-identity.md",
|
||||
"dnd.item_events/sharedassets/common-dnd-references.md",
|
||||
"dnd.item_events/sharedassets/common-dnd-system.md",
|
||||
"dnd.item_events/sharedassets/common-dnd-transcript.md",
|
||||
"dnd.item_events/task.md",
|
||||
}
|
||||
assertProductionContains(t, "production prompt assets", assetNames, requiredAssets)
|
||||
|
||||
catalog := catalogFromRegistries(registries)
|
||||
for _, test := range []struct {
|
||||
stage pipeline.ModuleStage
|
||||
key string
|
||||
want contracts.ExecutionClass
|
||||
}{
|
||||
{stage: pipeline.StageInput, key: "seriatim", want: contracts.ExecutionClassDeterministic},
|
||||
{stage: pipeline.StageChunk, key: "generic", want: contracts.ExecutionClassDeterministic},
|
||||
{stage: pipeline.StageChunk, key: "dnd/scenes", want: contracts.ExecutionClassLLMBacked},
|
||||
{stage: pipeline.StageExtract, key: "dnd/spells", want: contracts.ExecutionClassLLMBacked},
|
||||
{stage: pipeline.StageExtract, key: "dnd/npcs", want: contracts.ExecutionClassLLMBacked},
|
||||
{stage: pipeline.StageExtract, key: "dnd/combat-turns", want: contracts.ExecutionClassLLMBacked},
|
||||
{stage: pipeline.StageExtract, key: "dnd/item-events", want: contracts.ExecutionClassLLMBacked},
|
||||
{stage: pipeline.StageExtract, key: "dnd/npc-interactions", want: contracts.ExecutionClassLLMBacked},
|
||||
{stage: pipeline.StageExtract, key: "dnd/scene-descriptions", want: contracts.ExecutionClassLLMBacked},
|
||||
{stage: pipeline.StageMerge, key: "appendorder", want: contracts.ExecutionClassDeterministic},
|
||||
{stage: pipeline.StageNormalize, key: "noop", want: contracts.ExecutionClassDeterministic},
|
||||
{stage: pipeline.StageNormalize, key: "dnd/spells", want: contracts.ExecutionClassDeterministic},
|
||||
{stage: pipeline.StageNormalize, key: "dnd/npcs", want: contracts.ExecutionClassLLMBacked},
|
||||
{stage: pipeline.StageNormalize, key: "dnd/combat-turns", want: contracts.ExecutionClassDeterministic},
|
||||
{stage: pipeline.StageNormalize, key: "dnd/item-events", want: contracts.ExecutionClassDeterministic},
|
||||
{stage: pipeline.StageNormalize, key: "dnd/npc-interactions", want: contracts.ExecutionClassDeterministic},
|
||||
{stage: pipeline.StageNormalize, key: "dnd/scene-descriptions", want: contracts.ExecutionClassDeterministic},
|
||||
{stage: pipeline.StageOutput, key: "json", want: contracts.ExecutionClassDeterministic},
|
||||
} {
|
||||
got, ok := catalog.ExecutionClass(test.stage, test.key)
|
||||
if !ok || got != test.want {
|
||||
t.Fatalf("production execution class for %s/%s = %q, %t; want %q, true", test.stage, test.key, got, ok, test.want)
|
||||
}
|
||||
}
|
||||
converted := registriesFromCatalog(catalog)
|
||||
if converted.ArtifactCodecs != registries.ArtifactCodecs || converted.ValidatorChains != registries.ValidatorChains {
|
||||
t.Fatal("catalog/registry conversion did not preserve codec and validator-chain registries")
|
||||
if converted.ArtifactCodecs != registries.ArtifactCodecs || converted.ArtifactEvidence != registries.ArtifactEvidence || converted.ValidatorChains != registries.ValidatorChains {
|
||||
t.Fatal("catalog/registry conversion did not preserve artifact and validator registries")
|
||||
}
|
||||
codecSpec, ok := catalog.ArtifactCodecs.Spec(dnd.SpellListKind)
|
||||
if !ok || codecSpec.Kind != dnd.SpellListKind || codecSpec.Schema.ID != spellcodec.SchemaID {
|
||||
@@ -134,6 +207,10 @@ func TestProductionCatalogCoversMaintainedConfigurations(t *testing.T) {
|
||||
if !ok || combatCodecSpec.Kind != dnd.CombatTurnListKind || combatCodecSpec.Schema.ID != combatcodec.SchemaID {
|
||||
t.Fatalf("combat codec spec = %#v, ok=%t, want typed D&D combat codec", combatCodecSpec, ok)
|
||||
}
|
||||
itemEventCodecSpec, ok := catalog.ArtifactCodecs.Spec(dnd.ItemEventListKind)
|
||||
if !ok || itemEventCodecSpec.Kind != dnd.ItemEventListKind || itemEventCodecSpec.Schema.ID != itemeventcodec.SchemaID {
|
||||
t.Fatalf("item event codec spec = %#v, ok=%t, want typed D&D item-event codec", itemEventCodecSpec, ok)
|
||||
}
|
||||
if got := catalog.ValidatorChains.Validators(pipeline.StageExtract, spells.Key); !reflect.DeepEqual(got, wantChain) {
|
||||
t.Fatalf("catalog validator chain = %#v, want %#v", got, wantChain)
|
||||
}
|
||||
@@ -146,13 +223,83 @@ func TestProductionCatalogCoversMaintainedConfigurations(t *testing.T) {
|
||||
func TestDefaultCLICompositionValidatesRepresentativeConfiguration(t *testing.T) {
|
||||
var stdout, stderr strings.Builder
|
||||
code := RunWithOptions([]string{
|
||||
"config", "validate", "--config", repositoryPath("examples", "dnd-spells.config.yml"), "--pipeline", "dnd-session",
|
||||
"config", "validate", "--config", repositoryPath("examples", "dnd-minimal.config.yml"), "--pipeline", "dnd-session",
|
||||
}, &stdout, &stderr, Options{})
|
||||
if code != 0 || stderr.Len() != 0 {
|
||||
t.Fatalf("validate representative config with default composition: code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestProductionAssetsResolveDNDExtractionProfile(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
newEngine := func(profileFile string) (*promptkit.Engine, error) {
|
||||
t.Helper()
|
||||
options, err := components.assets.PromptKitOptions()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if profileFile != "" {
|
||||
options = append(options, promptkit.WithProfileFile(profileFile))
|
||||
}
|
||||
return promptkit.NewEngine(promptkit.Config{}, options...)
|
||||
}
|
||||
|
||||
t.Run("fallback", func(t *testing.T) {
|
||||
engine, err := newEngine("")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
inspection, err := engine.InspectProfile(context.Background(), "dnd-extraction")
|
||||
if err != nil {
|
||||
t.Fatalf("InspectProfile() error = %v, want fallback profile", err)
|
||||
}
|
||||
params := inspection.EffectiveModelParams
|
||||
if params.BackendID != "openrouter" || params.Model != "openai/gpt-5.6-luna" || params.TimeoutSeconds != 240 || params.ServiceTier != "flex" {
|
||||
t.Fatalf("fallback profile parameters = %#v", params)
|
||||
}
|
||||
if params.ReasoningEffort != "" || params.Temperature != 0 || params.MaxTokens != 0 || params.TopP != 0 {
|
||||
t.Fatalf("fallback profile selected optional provider controls: %#v", params)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("valid operator profile wins", func(t *testing.T) {
|
||||
profilePath := filepath.Join(t.TempDir(), "profiles.yaml")
|
||||
if err := os.WriteFile(profilePath, []byte(`id: dnd-extraction
|
||||
endpoint: http://operator.example.test/v1
|
||||
model: operator-model
|
||||
timeout_seconds: 75
|
||||
`), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
engine, err := newEngine(profilePath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
inspection, err := engine.InspectProfile(context.Background(), "dnd-extraction")
|
||||
if err != nil {
|
||||
t.Fatalf("InspectProfile() error = %v, want operator profile", err)
|
||||
}
|
||||
params := inspection.EffectiveModelParams
|
||||
if params.BackendID != "" || params.Endpoint != "http://operator.example.test/v1" || params.Model != "operator-model" || params.TimeoutSeconds != 75 || params.ServiceTier != "" {
|
||||
t.Fatalf("operator profile parameters = %#v, want complete replacement", params)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("invalid operator profile does not fall through", func(t *testing.T) {
|
||||
profilePath := filepath.Join(t.TempDir(), "profiles.yaml")
|
||||
if err := os.WriteFile(profilePath, []byte("id: dnd-extraction\nendpoint: http://operator.example.test/v1\nmodel: operator-model\nunknown: value\n"), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
engine, err := newEngine(profilePath)
|
||||
if err == nil {
|
||||
_, err = engine.InspectProfile(context.Background(), "dnd-extraction")
|
||||
}
|
||||
if err == nil {
|
||||
t.Fatal("operator profile error = nil, want failure instead of fallback")
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestProductionPromptAssetsPrepareWithoutProviderCredentials(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
cfg := config.Default()
|
||||
@@ -175,7 +322,7 @@ func TestProductionPromptAssetsPrepareWithoutProviderCredentials(t *testing.T) {
|
||||
|
||||
func TestProductionSpellValidatorsPrepareFromMaterializedCatalog(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
configPath := repositoryPath("examples", "dnd-spells-production.config.yml")
|
||||
configPath := writeProductionSpellCatalogContractConfig(t)
|
||||
effective, err := loadMaintainedExample(t, configPath).Resolve(resolveInputForMaintainedExample(components, "dnd-session"))
|
||||
if err != nil {
|
||||
t.Fatalf("resolve production spell configuration: %v", err)
|
||||
@@ -202,7 +349,7 @@ func TestProductionSpellValidatorsPrepareFromMaterializedCatalog(t *testing.T) {
|
||||
|
||||
func TestProductionSpellNormalizerRejectsInvalidCatalogReferencesBeforeExecution(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
configPath := repositoryPath("examples", "dnd-spells-production.config.yml")
|
||||
configPath := writeProductionSpellCatalogContractConfig(t)
|
||||
resolve := func(t *testing.T) pipeline.ResolvedPipeline {
|
||||
t.Helper()
|
||||
effective, err := loadMaintainedExample(t, configPath).Resolve(resolveInputForMaintainedExample(components, "dnd-session"))
|
||||
@@ -260,12 +407,15 @@ func TestProductionSpellNormalizerRejectsInvalidCatalogReferencesBeforeExecution
|
||||
t.Fatal(err)
|
||||
}
|
||||
checkpointRoot := filepath.Join(t.TempDir(), "checkpoints")
|
||||
content := string(readRepositoryFile(t, "examples", "dnd-spells-production.config.yml"))
|
||||
content = replaceRequiredOnce(t, content, "./dnd-spells-roster.txt", repositoryPath("examples", "dnd-spells-roster.txt"))
|
||||
content = replaceRequiredOnce(t, content, "./dnd-spells-glossary.txt", repositoryPath("examples", "dnd-spells-glossary.txt"))
|
||||
content = strings.Replace(content, "./dnd-spells-catalog.json", repositoryPath("examples", "dnd-spells-catalog.json"), 1)
|
||||
content = replaceRequiredOnce(t, content, "./dnd-spells-catalog.json", catalogPath)
|
||||
content = replaceRequiredOnce(t, content, " enabled: false\n directory: /var/cache/notarius/checkpoints", " enabled: true\n directory: "+checkpointRoot)
|
||||
content := productionSpellCatalogContractConfig(t)
|
||||
catalogSource := repositoryPath("examples", "dnd-spell-catalog.json")
|
||||
if count := strings.Count(content, catalogSource); count != 2 {
|
||||
t.Fatalf("spell catalog source occurs %d times, want extract and normalize bindings", count)
|
||||
}
|
||||
content = strings.Replace(content, catalogSource, "__extract_catalog__", 1)
|
||||
content = replaceRequiredOnce(t, content, catalogSource, catalogPath)
|
||||
content = replaceRequiredOnce(t, content, "__extract_catalog__", catalogSource)
|
||||
content = replaceRequiredOnce(t, content, " enabled: false\n directory: \"\"", " enabled: true\n directory: "+checkpointRoot)
|
||||
configFile := filepath.Join(t.TempDir(), "config.yml")
|
||||
if err := os.WriteFile(configFile, []byte(content), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
@@ -276,7 +426,7 @@ func TestProductionSpellNormalizerRejectsInvalidCatalogReferencesBeforeExecution
|
||||
options := Options{
|
||||
Catalog: catalogFromRegistries(components.registries),
|
||||
Registries: components.registries,
|
||||
LLMClientFactory: func(context.Context, config.Config, string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
LLMClientFactory: func(context.Context, config.Config, string, LLMRuntimeOverrides) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
llmConstructed = true
|
||||
return nil, nil, errors.New("LLM client must not be constructed")
|
||||
},
|
||||
@@ -324,18 +474,9 @@ func setNormalizeSpellCatalogSource(t *testing.T, resolved *pipeline.ResolvedPip
|
||||
resolved.Steps[0].ArtifactLanes[0].NormalizeReferences.Bindings = bindings
|
||||
}
|
||||
|
||||
func TestProductionLLMClientFactoriesBuildOfflineRuntime(t *testing.T) {
|
||||
func TestProductionLLMClientFactoryBuildsOfflineRuntime(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
factories := []struct {
|
||||
name string
|
||||
factory LLMClientFactory
|
||||
}{
|
||||
{name: "default production assets", factory: productionLLMClientFactory},
|
||||
{name: "provided production assets", factory: productionLLMClientFactoryWithAssets(components.assets)},
|
||||
}
|
||||
for _, tt := range factories {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
client, manifests, err := tt.factory(context.Background(), config.Default(), "test-profile")
|
||||
client, manifests, err := productionLLMClientFactoryWithAssets(components.assets)(context.Background(), config.Default(), "test-profile", LLMRuntimeOverrides{})
|
||||
if err != nil {
|
||||
t.Fatalf("build production LLM runtime: %v", err)
|
||||
}
|
||||
@@ -345,10 +486,138 @@ func TestProductionLLMClientFactoriesBuildOfflineRuntime(t *testing.T) {
|
||||
if len(manifests) != 0 {
|
||||
t.Fatalf("eager profile manifests = %#v, want none", manifests)
|
||||
}
|
||||
fingerprintProvider, ok := client.(llm.CheckpointFingerprintProvider)
|
||||
if !ok {
|
||||
t.Fatalf("production LLM client %T does not provide one profile-source checkpoint fingerprint", client)
|
||||
}
|
||||
fingerprints, err := fingerprintProvider.LLMCheckpointFingerprints()
|
||||
if err != nil || len(fingerprints) != 1 {
|
||||
t.Fatalf("production LLM checkpoint fingerprints = %#v, error = %v, want one profile-source identity", fingerprints, err)
|
||||
}
|
||||
if _, ok := client.(contracts.LLMProfileManifestProvider); !ok {
|
||||
t.Fatalf("production LLM client %T does not provide profile manifests", client)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeOptionsSharesProductionProfileAssetsWithDefaultRuntime(t *testing.T) {
|
||||
opts, err := normalizeOptions(Options{
|
||||
Catalog: pipeline.ModuleCatalog{Inputs: pipeline.NewInputAdapterRegistry()},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if opts.promptKitAssets == nil || opts.LLMClientFactory == nil {
|
||||
t.Fatalf("normalized options = %#v, want shared profile assets and default runtime factory", opts)
|
||||
}
|
||||
if err := validateExplicitPromptKitProfiles(context.Background(), config.Default(), []string{"dnd-extraction"}, opts.promptKitAssets); err != nil {
|
||||
t.Fatalf("inspect application fallback profile: %v", err)
|
||||
}
|
||||
|
||||
client, _, err := opts.LLMClientFactory(context.Background(), config.Default(), "dnd-extraction", LLMRuntimeOverrides{})
|
||||
if err != nil {
|
||||
t.Fatalf("build default runtime: %v", err)
|
||||
}
|
||||
fingerprintProvider, ok := client.(llm.CheckpointFingerprintProvider)
|
||||
if !ok {
|
||||
t.Fatalf("default runtime client %T does not provide checkpoint fingerprints", client)
|
||||
}
|
||||
runtimeFingerprints, err := fingerprintProvider.LLMCheckpointFingerprints()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
directClient, err := llm.NewPromptKitClient(llm.PromptKitClientConfig{Assets: opts.promptKitAssets})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
inspectionFingerprints, err := directClient.LLMCheckpointFingerprints()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !reflect.DeepEqual(runtimeFingerprints, inspectionFingerprints) {
|
||||
t.Fatalf("runtime profile fingerprints = %#v, inspection profile fingerprints = %#v", runtimeFingerprints, inspectionFingerprints)
|
||||
}
|
||||
}
|
||||
|
||||
func TestProductionLLMClientFactoryUsesConfiguredLocalBackend(t *testing.T) {
|
||||
var providerCalls atomic.Int32
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
providerCalls.Add(1)
|
||||
if r.URL.Path != "/v1/chat/completions" {
|
||||
t.Errorf("provider path = %q, want /v1/chat/completions", r.URL.Path)
|
||||
}
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
_, _ = w.Write([]byte(`{
|
||||
"choices": [{"message": {"role": "assistant", "content": "{\"ok\":true}"}}],
|
||||
"usage": {"prompt_tokens": 3, "completion_tokens": 4, "total_tokens": 7}
|
||||
}`))
|
||||
}))
|
||||
defer server.Close()
|
||||
|
||||
profilePath := filepath.Join(t.TempDir(), "profiles.yml")
|
||||
if err := os.WriteFile(profilePath, []byte(`id: local-profile
|
||||
backend: local
|
||||
model: local-model
|
||||
`), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
assets := llm.NewAssetRegistry()
|
||||
if err := assets.RegisterPromptFS(fstest.MapFS{
|
||||
"production.local.yaml": {Data: []byte(`id: production.local
|
||||
version: "v1"
|
||||
inputs:
|
||||
- name: transcript
|
||||
required: true
|
||||
messages:
|
||||
- role: user
|
||||
content: '{{ input "transcript" }}'
|
||||
output:
|
||||
format: json
|
||||
validation_mode: json
|
||||
`)},
|
||||
}, "."); err != nil {
|
||||
t.Fatalf("register prompt assets: %v", err)
|
||||
}
|
||||
|
||||
cfg := config.Default()
|
||||
cfg.PromptKit.ProfileFile = profilePath
|
||||
cfg.PromptKit.LocalBackend = &config.PromptKitLocalBackendConfig{
|
||||
Endpoint: server.URL + "/v1",
|
||||
ConcurrencyLimit: 2,
|
||||
}
|
||||
client, manifests, err := productionLLMClientFactoryWithAssets(assets)(
|
||||
context.Background(),
|
||||
cfg,
|
||||
"local-profile",
|
||||
LLMRuntimeOverrides{},
|
||||
)
|
||||
if err != nil {
|
||||
t.Fatalf("build production LLM runtime: %v", err)
|
||||
}
|
||||
if len(manifests) != 0 {
|
||||
t.Fatalf("eager profile manifests = %#v, want none", manifests)
|
||||
}
|
||||
|
||||
var out map[string]any
|
||||
_, err = client.CompleteStructured(context.Background(), contracts.StructuredCompletionRequest{
|
||||
PromptID: "production.local",
|
||||
ProfileID: "local-profile",
|
||||
Inputs: contracts.LLMInputSet{
|
||||
"transcript": contracts.NewLLMInputMaterial("transcript", "text/plain", []byte("local request"), "", ""),
|
||||
},
|
||||
}, &out)
|
||||
if err != nil {
|
||||
t.Fatalf("CompleteStructured() error = %v, want nil", err)
|
||||
}
|
||||
if providerCalls.Load() != 1 {
|
||||
t.Fatalf("provider calls = %d, want 1", providerCalls.Load())
|
||||
}
|
||||
provider, ok := client.(contracts.LLMProfileManifestProvider)
|
||||
if !ok {
|
||||
t.Fatalf("production client %T does not provide profile manifests", client)
|
||||
}
|
||||
recorded := provider.LLMProfileManifests()
|
||||
if len(recorded) != 1 || recorded[0].BackendID != promptkit.BackendLocal {
|
||||
t.Fatalf("production profile manifests = %#v, want local backend", recorded)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -356,14 +625,15 @@ func TestProductionLLMClientFactoriesRejectInvalidConstruction(t *testing.T) {
|
||||
t.Run("canceled context", func(t *testing.T) {
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
cancel()
|
||||
client, manifests, err := productionLLMClientFactory(ctx, config.Default(), "test-profile")
|
||||
components := productionTestComponents(t)
|
||||
client, manifests, err := productionLLMClientFactoryWithAssets(components.assets)(ctx, config.Default(), "test-profile", LLMRuntimeOverrides{})
|
||||
if !errors.Is(err, context.Canceled) || client != nil || len(manifests) != 0 {
|
||||
t.Fatalf("client=%T manifests=%#v error=%v, want canceled construction", client, manifests, err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("nil assets", func(t *testing.T) {
|
||||
client, manifests, err := productionLLMClientFactoryWithAssets(nil)(context.Background(), config.Default(), "test-profile")
|
||||
client, manifests, err := productionLLMClientFactoryWithAssets(nil)(context.Background(), config.Default(), "test-profile", LLMRuntimeOverrides{})
|
||||
if err == nil || !strings.Contains(err.Error(), "asset registry must not be nil") || client != nil || len(manifests) != 0 {
|
||||
t.Fatalf("client=%T manifests=%#v error=%v, want nil-assets failure", client, manifests, err)
|
||||
}
|
||||
@@ -373,7 +643,7 @@ func TestProductionLLMClientFactoriesRejectInvalidConstruction(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
cfg := config.Default()
|
||||
cfg.Concurrency.TotalLLM = 0
|
||||
client, manifests, err := productionLLMClientFactoryWithAssets(components.assets)(context.Background(), cfg, "test-profile")
|
||||
client, manifests, err := productionLLMClientFactoryWithAssets(components.assets)(context.Background(), cfg, "test-profile", LLMRuntimeOverrides{})
|
||||
if err == nil || !strings.Contains(err.Error(), "create LLM scheduler") || !strings.Contains(err.Error(), "greater than zero") || client != nil || len(manifests) != 0 {
|
||||
t.Fatalf("client=%T manifests=%#v error=%v, want scheduler-construction failure", client, manifests, err)
|
||||
}
|
||||
@@ -381,7 +651,7 @@ func TestProductionLLMClientFactoriesRejectInvalidConstruction(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestProductionConfigValidationCoversModuleAndVariantFailures(t *testing.T) {
|
||||
base := string(readRepositoryFile(t, "examples", "dnd-spells.config.yml"))
|
||||
base := string(readRepositoryFile(t, "examples", "dnd-minimal.config.yml"))
|
||||
validPath := writeProductionContractConfig(t, base)
|
||||
options := productionCLIOptions(t)
|
||||
var stdout, stderr strings.Builder
|
||||
@@ -437,8 +707,8 @@ func TestProductionConfigValidationCoversModuleAndVariantFailures(t *testing.T)
|
||||
}
|
||||
|
||||
func TestProductionNormalizeValidatorOverrideRemainsAuthoritative(t *testing.T) {
|
||||
base := string(readRepositoryFile(t, "examples", "dnd-spells.config.yml"))
|
||||
content := replaceRequiredOnce(t, base, " normalize: dnd/spells\n", " normalize:\n module: dnd/spells\n validators:\n - module: generic/always_accept\n")
|
||||
base := string(readRepositoryFile(t, "examples", "dnd-minimal.config.yml"))
|
||||
content := replaceRequiredOnce(t, base, " normalize: dnd/spells\n", " normalize:\n module: dnd/spells\n validators:\n - module: generic/always_accept\n - module: generic/valid_json\n")
|
||||
path := writeProductionContractConfig(t, content)
|
||||
components := productionTestComponents(t)
|
||||
effective, err := loadMaintainedExample(t, path).Resolve(resolveInputForMaintainedExample(components, "dnd-session"))
|
||||
@@ -449,15 +719,15 @@ func TestProductionNormalizeValidatorOverrideRemainsAuthoritative(t *testing.T)
|
||||
if chain.Stage != pipeline.StageNormalize || chain.ModuleKey != spellnormalize.Key {
|
||||
continue
|
||||
}
|
||||
if len(chain.Validators) != 1 || chain.Validators[0].Binding.Module != "generic/always_accept" {
|
||||
t.Fatalf("normalize validator chain = %#v, want explicit always-accept override", chain)
|
||||
if len(chain.Validators) != 2 || chain.Validators[0].Binding.Module != "generic/always_accept" || chain.Validators[1].Binding.Module != "generic/valid_json" {
|
||||
t.Fatalf("normalize validator chain = %#v, want explicit validator order", chain)
|
||||
}
|
||||
return
|
||||
}
|
||||
t.Fatalf("resolved validator chains = %#v, want normalize chain for %q", effective.ResolvedPipeline.ValidatorChains, spellnormalize.Key)
|
||||
}
|
||||
|
||||
func TestProductionSceneRunRecordsChunkerWarningsAndProvenance(t *testing.T) {
|
||||
func TestProductionSceneRunRecordsAnnotationFreeChunkPlanAndProvenance(t *testing.T) {
|
||||
outputRoot := filepath.Join(t.TempDir(), "output")
|
||||
configPath := writeProductionContractConfig(t, productionRunConfig(outputRoot, "dnd/scenes"))
|
||||
fake := &productionFakeLLMClient{}
|
||||
@@ -481,34 +751,95 @@ func TestProductionSceneRunRecordsChunkerWarningsAndProvenance(t *testing.T) {
|
||||
if got := manifest.ChunkPlan.ProducerMetadata["response_schema_id"]; got != scenes.ResponseSchemaID {
|
||||
t.Fatalf("chunk producer schema metadata = %#v, want %q", got, scenes.ResponseSchemaID)
|
||||
}
|
||||
index := readProductionJSON[productionChunkMapIndex](t, filepath.Join(outputRoot, productionRunID, "index.json"))
|
||||
if index.ChunkMap == nil || index.ChunkMap.ArtifactKind != chunkmap.ArtifactKind || index.ChunkMap.File != "chunk-map.json" || index.ChunkMap.MediaType != chunkmap.MediaType || index.ChunkMap.SchemaID != chunkmap.SchemaID || index.ChunkMap.SchemaName != chunkmap.SchemaName || index.ChunkMap.SchemaVersion != chunkmap.SchemaVersion {
|
||||
t.Fatalf("chunk map index = %#v, want fixed chunk map descriptor", index.ChunkMap)
|
||||
}
|
||||
for _, output := range index.OutputFiles {
|
||||
if output.File == index.ChunkMap.File {
|
||||
t.Fatalf("lane output files = %#v, want no chunk map", index.OutputFiles)
|
||||
}
|
||||
}
|
||||
content, err := os.ReadFile(filepath.Join(outputRoot, productionRunID, index.ChunkMap.File))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
chunkMap, err := chunkmap.New().Decode(content)
|
||||
if err != nil {
|
||||
t.Fatalf("Decode(chunk map) error = %v", err)
|
||||
}
|
||||
if chunkMap.SourceID != "session-alpha" || chunkMap.SourceDigest != manifest.ChunkPlan.SourceDigest || chunkMap.PlanDigest != manifest.ChunkPlan.PlanDigest || chunkMap.RequestedChunker != scenes.Key || chunkMap.Producer.InputModule != "seriatim" || chunkMap.Producer.ChunkModule != scenes.Key || chunkMap.Producer.LLMProfile != manifest.ChunkPlan.ProducerLLMProfile {
|
||||
t.Fatalf("chunk map identity and producer = %#v, want accepted scene plan provenance", chunkMap)
|
||||
}
|
||||
if len(chunkMap.Chunks) != 1 || chunkMap.Chunks[0].ID != "chunk-000001" || chunkMap.Chunks[0].Index != 0 || chunkMap.Chunks[0].SourceRef.SourceID != "session-alpha" || chunkMap.Chunks[0].SourceRef.StartUnitID != 1 || chunkMap.Chunks[0].SourceRef.EndUnitID != 2 || chunkMap.Chunks[0].UnitCount != 2 {
|
||||
t.Fatalf("chunk map chunks = %#v, want one stable accepted scene range", chunkMap.Chunks)
|
||||
}
|
||||
if len(chunkMap.PlanAnnotations) != 0 {
|
||||
t.Fatalf("chunk map plan annotations = %#v, want none", chunkMap.PlanAnnotations)
|
||||
}
|
||||
if len(chunkMap.Chunks[0].Annotations) != 0 {
|
||||
t.Fatalf("chunk map range annotations = %#v, want none", chunkMap.Chunks[0].Annotations)
|
||||
}
|
||||
warnings := readProductionJSON[struct {
|
||||
Warnings []contracts.Warning `json:"warnings"`
|
||||
}](t, filepath.Join(outputRoot, productionRunID, "warnings.json"))
|
||||
if len(warnings.Warnings) != 1 || warnings.Warnings[0].ReasonCode != "scene_boundary_caveat" {
|
||||
t.Fatalf("warnings = %#v, want one scene boundary warning", warnings.Warnings)
|
||||
if len(warnings.Warnings) != 0 {
|
||||
t.Fatalf("warnings = %#v, want none", warnings.Warnings)
|
||||
}
|
||||
if len(fake.requestsFor(scenes.PromptID)) != 1 || len(fake.requestsFor(spells.PromptID)) != 1 {
|
||||
t.Fatalf("fake prompt requests = %#v, want one scene and one spell request", fake.requestPrompts())
|
||||
if len(fake.requestsFor(scenes.PromptID)) != 1 || len(fake.requestsFor(spells.PromptID)) != 1 || len(fake.requestsFor(itemeventextract.PromptID)) != 1 {
|
||||
t.Fatalf("fake prompt requests = %#v, want one scene, spell, and item-event request", fake.requestPrompts())
|
||||
}
|
||||
}
|
||||
|
||||
type maintainedExample struct {
|
||||
name string
|
||||
path string
|
||||
transcriptPath string
|
||||
pipelineIDs []string
|
||||
}
|
||||
|
||||
func maintainedExampleFiles(t *testing.T) []maintainedExample {
|
||||
t.Helper()
|
||||
return []maintainedExample{
|
||||
{name: "minimal", path: repositoryPath("examples", "dnd-spells.config.yml"), pipelineIDs: []string{"dnd-session"}},
|
||||
{name: "production", path: repositoryPath("examples", "dnd-spells-production.config.yml"), pipelineIDs: []string{"dnd-session"}},
|
||||
{name: "npcs", path: repositoryPath("examples", "dnd-npcs.config.yml"), pipelineIDs: []string{"dnd-session"}},
|
||||
{name: "combat", path: repositoryPath("examples", "dnd-combat-turns.config.yml"), pipelineIDs: []string{"dnd-combat"}},
|
||||
{name: "npc-grounded", path: repositoryPath("examples", "dnd-npc-grounded.config.yml"), pipelineIDs: []string{"dnd-npc-grounded"}},
|
||||
{name: "minimal", path: repositoryPath("examples", "dnd-minimal.config.yml"), transcriptPath: repositoryPath("examples", "seriatim-minimal-transcript.json"), pipelineIDs: []string{"dnd-session"}},
|
||||
{name: "complete", path: repositoryPath("examples", "dnd-complete.config.yml"), transcriptPath: repositoryPath("examples", "dnd-complete-transcript.json"), pipelineIDs: []string{"dnd-session"}},
|
||||
}
|
||||
}
|
||||
|
||||
func productionSpellCatalogContractConfig(t *testing.T) string {
|
||||
t.Helper()
|
||||
return fmt.Sprintf(`version: 4
|
||||
cache:
|
||||
chunk_plans:
|
||||
mode: bypass
|
||||
checkpoints:
|
||||
enabled: false
|
||||
directory: ""
|
||||
pipelines:
|
||||
dnd-session:
|
||||
input: seriatim
|
||||
references:
|
||||
party: %q
|
||||
glossary: %q
|
||||
artifacts:
|
||||
spells:
|
||||
extract:
|
||||
module: dnd/spells
|
||||
retries: 2
|
||||
references:
|
||||
spell_catalog: %q
|
||||
normalize:
|
||||
module: dnd/spells
|
||||
references:
|
||||
spell_catalog: %q
|
||||
`, repositoryPath("examples", "dnd-party.txt"), repositoryPath("examples", "dnd-glossary.txt"), repositoryPath("examples", "dnd-spell-catalog.json"), repositoryPath("examples", "dnd-spell-catalog.json"))
|
||||
}
|
||||
|
||||
func writeProductionSpellCatalogContractConfig(t *testing.T) string {
|
||||
t.Helper()
|
||||
return writeProductionContractConfig(t, productionSpellCatalogContractConfig(t))
|
||||
}
|
||||
|
||||
func loadMaintainedExample(t *testing.T, path string) config.Config {
|
||||
t.Helper()
|
||||
fileConfig, err := config.LoadFileConfig(path)
|
||||
@@ -545,6 +876,7 @@ func productionOptionsFromComponents(components productionComponents) Options {
|
||||
Catalog: catalogFromRegistries(components.registries),
|
||||
Registries: components.registries,
|
||||
LookupEnv: emptyLookup,
|
||||
promptKitAssets: components.assets,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -567,14 +899,14 @@ func productionRunOptions(t *testing.T, fake *productionFakeLLMClient) Options {
|
||||
options.Now = func() time.Time { return time.Unix(1700000000, 0).UTC() }
|
||||
options.RunIDGenerator = func(time.Time) (string, error) { return productionRunID, nil }
|
||||
options.UserCacheDir = func() (string, error) { return "", errors.New("user cache must not be used") }
|
||||
options.LLMClientFactory = func(context.Context, config.Config, string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
options.LLMClientFactory = func(context.Context, config.Config, string, LLMRuntimeOverrides) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
return fake, nil, nil
|
||||
}
|
||||
return options
|
||||
}
|
||||
|
||||
func productionRunConfig(outputRoot, chunkModule string) string {
|
||||
return fmt.Sprintf(`version: 3
|
||||
return fmt.Sprintf(`version: 4
|
||||
output:
|
||||
directory: %q
|
||||
cache:
|
||||
@@ -587,12 +919,32 @@ pipelines:
|
||||
dnd-session:
|
||||
input: seriatim
|
||||
chunk: %s
|
||||
output:
|
||||
module: json
|
||||
options:
|
||||
include_chunk_map: true
|
||||
artifacts:
|
||||
spells:
|
||||
extract: dnd/spells
|
||||
item-events:
|
||||
extract: dnd/item-events
|
||||
`, outputRoot, filepath.Join(filepath.Dir(outputRoot), "debug"), chunkModule)
|
||||
}
|
||||
|
||||
type productionChunkMapIndex struct {
|
||||
OutputFiles []struct {
|
||||
File string `json:"file"`
|
||||
} `json:"output_files"`
|
||||
ChunkMap *struct {
|
||||
ArtifactKind contracts.ArtifactKind `json:"artifact_kind"`
|
||||
File string `json:"file"`
|
||||
MediaType string `json:"media_type"`
|
||||
SchemaID string `json:"schema_id"`
|
||||
SchemaName string `json:"schema_name"`
|
||||
SchemaVersion string `json:"schema_version"`
|
||||
} `json:"chunk_map"`
|
||||
}
|
||||
|
||||
func writeProductionContractConfig(t *testing.T, content string) string {
|
||||
t.Helper()
|
||||
path := filepath.Join(t.TempDir(), "config.yml")
|
||||
@@ -667,13 +1019,15 @@ func (client *productionFakeLLMClient) CompleteStructured(ctx context.Context, r
|
||||
var content []byte
|
||||
switch req.PromptID {
|
||||
case scenes.PromptID:
|
||||
content = []byte(`{"scenes":[{"start_unit_id":1,"end_unit_id":2,"short_title":"Opening scene","primary_mode":"Narrative","main_participants":["Aria"],"summary":"The session opens.","boundary_note":"The opening covers the available transcript.","boundary_confidence":"High"}],"boundary_caveats":["The opening boundary is inferred from the short transcript."]}`)
|
||||
content = []byte(`{"scenes":[{"start_unit_id":1,"end_unit_id":2}]}`)
|
||||
case spells.PromptID:
|
||||
if client.spellResponse != "" {
|
||||
content = []byte(client.spellResponse)
|
||||
} else {
|
||||
content = []byte(`{"spell_casts":[{"caster":"Aria","spell":"Cure Wounds","effect":"Heals an injured ally.","narrative_description":"Aria restores the fighter after the fight.","source_refs":[{"source_id":"session-alpha","start_unit_id":1,"end_unit_id":1}]}]}`)
|
||||
content = []byte(`{"spell_casts":[{"caster":"Aria","spell":"Cure Wounds","source_refs":[{"start_unit_id":1,"end_unit_id":1}]}]}`)
|
||||
}
|
||||
case itemeventextract.PromptID:
|
||||
content = []byte(`{"events":[{"name":"Cure Wounds","kind":"acquired","to":"party","source_refs":[{"start_segment":1,"end_segment":1}]}]}`)
|
||||
default:
|
||||
return contracts.StructuredCompletionResponse{}, fmt.Errorf("unexpected prompt %q", req.PromptID)
|
||||
}
|
||||
|
||||
46
internal/cli/promptkit_profiles.go
Normal file
46
internal/cli/promptkit_profiles.go
Normal file
@@ -0,0 +1,46 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
)
|
||||
|
||||
func validateExplicitPromptKitProfiles(ctx context.Context, cfg config.Config, profileIDs []string, assets *llm.AssetRegistry) error {
|
||||
if len(profileIDs) == 0 {
|
||||
return nil
|
||||
}
|
||||
inspector, err := llm.NewPromptKitProfileInspector(llm.PromptKitProfileInspectorConfig{
|
||||
Source: promptKitProfileSourceConfig(cfg),
|
||||
Assets: assets,
|
||||
})
|
||||
if err != nil {
|
||||
return fmt.Errorf("load PromptKit profiles: %w", err)
|
||||
}
|
||||
for _, profileID := range profileIDs {
|
||||
if _, err := inspector.InspectProfile(ctx, profileID); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func promptKitProfileSourceConfig(cfg config.Config) llm.PromptKitProfileSourceConfig {
|
||||
return llm.PromptKitProfileSourceConfig{
|
||||
ProfileDir: cfg.PromptKit.ProfileDir,
|
||||
ProfileFile: cfg.PromptKit.ProfileFile,
|
||||
LocalBackend: mapPromptKitLocalBackend(cfg.PromptKit.LocalBackend),
|
||||
}
|
||||
}
|
||||
|
||||
func mapPromptKitLocalBackend(cfg *config.PromptKitLocalBackendConfig) *llm.PromptKitLocalBackendConfig {
|
||||
if cfg == nil {
|
||||
return nil
|
||||
}
|
||||
return &llm.PromptKitLocalBackendConfig{
|
||||
Endpoint: cfg.Endpoint,
|
||||
ConcurrencyLimit: cfg.ConcurrencyLimit,
|
||||
}
|
||||
}
|
||||
159
internal/cli/promptkit_profiles_test.go
Normal file
159
internal/cli/promptkit_profiles_test.go
Normal file
@@ -0,0 +1,159 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"testing/fstest"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
)
|
||||
|
||||
func TestExplicitPromptKitProfileValidationInspectsProfilesWithoutGeneration(t *testing.T) {
|
||||
var providerCalls atomic.Int32
|
||||
server := httptest.NewServer(http.HandlerFunc(func(http.ResponseWriter, *http.Request) {
|
||||
providerCalls.Add(1)
|
||||
}))
|
||||
defer server.Close()
|
||||
|
||||
writeProfile := func(t *testing.T, name, content string) string {
|
||||
t.Helper()
|
||||
profilePath := filepath.Join(t.TempDir(), name+".yaml")
|
||||
if err := os.WriteFile(profilePath, []byte(content), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return profilePath
|
||||
}
|
||||
localProfile := "id: local-profile\nbackend: local\nmodel: local-model\n"
|
||||
credentialProfile := `id: credential-profile
|
||||
endpoint: ` + server.URL + `/v1
|
||||
model: credential-model
|
||||
api_key_env: NOTARIUS_PROMPTKIT_PROFILE_INSPECTION_TEST_KEY
|
||||
`
|
||||
t.Setenv("NOTARIUS_PROMPTKIT_PROFILE_INSPECTION_TEST_KEY", "")
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
profilePath string
|
||||
profileID string
|
||||
profileDir bool
|
||||
localBackend bool
|
||||
canceled bool
|
||||
wantErr []string
|
||||
rejectErr []string
|
||||
}{
|
||||
{
|
||||
name: "configured local backend",
|
||||
profilePath: writeProfile(t, "local-profile", localProfile),
|
||||
profileID: "local-profile",
|
||||
profileDir: true,
|
||||
localBackend: true,
|
||||
},
|
||||
{
|
||||
name: "missing local backend registration",
|
||||
profilePath: writeProfile(t, "local-profile", localProfile),
|
||||
profileID: "local-profile",
|
||||
wantErr: []string{`PromptKit profile "local-profile" is invalid or unreadable`},
|
||||
},
|
||||
{
|
||||
name: "absent profile",
|
||||
profilePath: writeProfile(t, "local-profile", localProfile),
|
||||
profileID: "absent-profile",
|
||||
localBackend: true,
|
||||
wantErr: []string{`PromptKit profile "absent-profile" is not configured`},
|
||||
},
|
||||
{
|
||||
name: "malformed profile",
|
||||
profilePath: writeProfile(t, "malformed-profile", "id: malformed-profile\nbackend: [\n"),
|
||||
profileID: "malformed-profile",
|
||||
wantErr: []string{`PromptKit profile "malformed-profile" is invalid or unreadable`},
|
||||
rejectErr: []string{"malformed-profile.yaml", "backend: ["},
|
||||
},
|
||||
{
|
||||
name: "invalid profile source",
|
||||
profilePath: filepath.Join(t.TempDir(), "missing-profile.yaml"),
|
||||
profileID: "missing-profile",
|
||||
wantErr: []string{"load PromptKit profiles", "profile configuration is invalid or unreadable"},
|
||||
},
|
||||
{
|
||||
name: "credential environment intentionally unset",
|
||||
profilePath: writeProfile(t, "credential-profile", credentialProfile),
|
||||
profileID: "credential-profile",
|
||||
},
|
||||
{
|
||||
name: "canceled inspection",
|
||||
profilePath: writeProfile(t, "local-profile", localProfile),
|
||||
profileID: "local-profile",
|
||||
localBackend: true,
|
||||
canceled: true,
|
||||
wantErr: []string{"context canceled"},
|
||||
},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
cfg := config.Default()
|
||||
if tt.profileDir {
|
||||
cfg.PromptKit.ProfileDir = filepath.Dir(tt.profilePath)
|
||||
} else {
|
||||
cfg.PromptKit.ProfileFile = tt.profilePath
|
||||
}
|
||||
if tt.localBackend {
|
||||
cfg.PromptKit.LocalBackend = &config.PromptKitLocalBackendConfig{
|
||||
Endpoint: server.URL + "/v1",
|
||||
ConcurrencyLimit: 2,
|
||||
}
|
||||
}
|
||||
ctx := context.Background()
|
||||
if tt.canceled {
|
||||
var cancel context.CancelFunc
|
||||
ctx, cancel = context.WithCancel(ctx)
|
||||
cancel()
|
||||
}
|
||||
err := validateExplicitPromptKitProfiles(ctx, cfg, []string{tt.profileID}, nil)
|
||||
if len(tt.wantErr) == 0 {
|
||||
if err != nil {
|
||||
t.Fatalf("validateExplicitPromptKitProfiles() error = %v, want nil", err)
|
||||
}
|
||||
return
|
||||
}
|
||||
if err == nil {
|
||||
t.Fatal("validateExplicitPromptKitProfiles() error = nil, want failure")
|
||||
}
|
||||
if tt.canceled && !errors.Is(err, context.Canceled) {
|
||||
t.Fatalf("canceled inspection error = %v, want context canceled", err)
|
||||
}
|
||||
for _, want := range tt.wantErr {
|
||||
if !strings.Contains(err.Error(), want) {
|
||||
t.Fatalf("validation error = %q, want %q", err, want)
|
||||
}
|
||||
}
|
||||
for _, rejected := range append(tt.rejectErr, tt.profilePath) {
|
||||
if rejected != "" && strings.Contains(err.Error(), rejected) {
|
||||
t.Fatalf("validation error = %q, must not expose %q", err, rejected)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
if providerCalls.Load() != 0 {
|
||||
t.Fatalf("provider calls during profile inspection = %d, want 0", providerCalls.Load())
|
||||
}
|
||||
}
|
||||
|
||||
func TestExplicitPromptKitProfileValidationUsesFallbackAssets(t *testing.T) {
|
||||
assets := llm.NewAssetRegistry()
|
||||
if err := assets.RegisterFallbackProfileFS(fstest.MapFS{
|
||||
"profiles/fallback.yaml": {Data: []byte("id: fallback-profile\nendpoint: http://promptkit.test/v1\nmodel: fallback-model\n")},
|
||||
}, "profiles"); err != nil {
|
||||
t.Fatalf("RegisterFallbackProfileFS() error = %v, want nil", err)
|
||||
}
|
||||
if err := validateExplicitPromptKitProfiles(context.Background(), config.Default(), []string{"fallback-profile"}, assets); err != nil {
|
||||
t.Fatalf("validateExplicitPromptKitProfiles() error = %v, want nil", err)
|
||||
}
|
||||
}
|
||||
@@ -116,7 +116,7 @@ func (h *recomputeTestHarness) options() Options {
|
||||
for _, key := range []string{"test/extract/producer", "test/extract/unrelated", "test/extract/middle", "test/extract/dependent"} {
|
||||
moduleKey := key
|
||||
spec := pipeline.ModuleSpec{
|
||||
Key: moduleKey, Stage: pipeline.StageExtract, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: stateTestArtifactKind,
|
||||
Key: moduleKey, Stage: pipeline.StageExtract, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: stateTestArtifactKind,
|
||||
ReferenceSlots: []contracts.ReferenceSlot{{Name: "upstream", AcceptedMediaTypes: []string{"application/json"}, AcceptedArtifactKinds: []contracts.ArtifactKind{stateTestArtifactKind}}},
|
||||
}
|
||||
if err := pipeline.RegisterExtractor(opts.Registries.Extractors, spec, func() (contracts.Extractor[stateTestArtifact], error) {
|
||||
@@ -125,7 +125,7 @@ func (h *recomputeTestHarness) options() Options {
|
||||
panic(err)
|
||||
}
|
||||
}
|
||||
if err := opts.Registries.Outputs.RegisterWithSpec(pipeline.ModuleSpec{Key: "test/recompute-output", Stage: pipeline.StageOutput, Requires: []string{"normalized"}, Provides: []string{"output"}}, func() (contracts.OutputEncoder, error) {
|
||||
if err := opts.Registries.Outputs.RegisterWithSpec(pipeline.ModuleSpec{Key: "test/recompute-output", Stage: pipeline.StageOutput, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"normalized"}, Provides: []string{"output"}}, func() (contracts.OutputEncoder, error) {
|
||||
return recomputeTestOutput{}, nil
|
||||
}); err != nil {
|
||||
panic(err)
|
||||
@@ -194,7 +194,7 @@ func (recomputeTestOutput) Encode(_ context.Context, req contracts.OutputRequest
|
||||
func newRecomputeTestRoots(t *testing.T) stateTestRoots {
|
||||
t.Helper()
|
||||
roots := newStateTestRoots(t)
|
||||
config := fmt.Sprintf(`version: 3
|
||||
config := fmt.Sprintf(`version: 4
|
||||
output:
|
||||
directory: %q
|
||||
cache:
|
||||
|
||||
@@ -217,7 +217,7 @@ func TestReferenceMaterializationSeparatesCLIAndConfigPathOrigins(t *testing.T)
|
||||
workingDir := t.TempDir()
|
||||
cfg := referenceContractConfig()
|
||||
configPath := filepath.Join(configDir, "config.yml")
|
||||
if err := os.WriteFile(configPath, []byte("version: 3\n"), 0o600); err != nil {
|
||||
if err := os.WriteFile(configPath, []byte("version: 4\n"), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(configDir, "required.txt"), []byte("config reference"), 0o600); err != nil {
|
||||
@@ -364,21 +364,21 @@ func referenceContractCatalog(t *testing.T, includeBetaMerger, includeBetaNormal
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
register(registries.Inputs.RegisterBuilderWithSpec(pipeline.ModuleSpec{Key: "reference/input", Stage: pipeline.StageInput, Provides: []string{"source"}}, func(map[string]any) error { return nil }, func(pipeline.BuildRequest) (contracts.InputAdapter, error) { return stateTestInput{}, nil }))
|
||||
register(registries.Chunkers.RegisterWithSpec(pipeline.ModuleSpec{Key: "reference/chunk", Stage: pipeline.StageChunk, Requires: []string{"source"}, Provides: []string{"chunks"}, ReferenceSlots: []contracts.ReferenceSlot{{Name: "chunk-slot"}, {Name: "required-chunk", Required: true}}}, func() (contracts.Chunker, error) { return stateTestChunker{}, nil }))
|
||||
register(registries.Inputs.RegisterBuilderWithSpec(pipeline.ModuleSpec{Key: "reference/input", Stage: pipeline.StageInput, ExecutionClass: contracts.ExecutionClassDeterministic, Provides: []string{"source"}}, func(map[string]any) error { return nil }, func(pipeline.BuildRequest) (contracts.InputAdapter, error) { return stateTestInput{}, nil }))
|
||||
register(registries.Chunkers.RegisterWithSpec(pipeline.ModuleSpec{Key: "reference/chunk", Stage: pipeline.StageChunk, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"source"}, Provides: []string{"chunks"}, ReferenceSlots: []contracts.ReferenceSlot{{Name: "chunk-slot"}, {Name: "required-chunk", Required: true}}}, func() (contracts.Chunker, error) { return stateTestChunker{}, nil }))
|
||||
register(pipeline.RegisterArtifactCodec(registries.ArtifactCodecs, referenceContractCodecA{}))
|
||||
register(pipeline.RegisterArtifactCodec(registries.ArtifactCodecs, referenceContractCodecB{}))
|
||||
register(pipeline.RegisterExtractor(registries.Extractors, pipeline.ModuleSpec{Key: "reference/extract-alpha", Stage: pipeline.StageExtract, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: referenceContractKindAlpha, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "alpha-slot"}, {Name: "required-extract", Required: true}}}, func() (contracts.Extractor[stateTestArtifact], error) { return stateTestExtractor{}, nil }))
|
||||
register(pipeline.RegisterExtractor(registries.Extractors, pipeline.ModuleSpec{Key: "reference/extract-beta", Stage: pipeline.StageExtract, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: referenceContractKindBeta, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "beta-slot"}, {Name: "required-extract", Required: true}}}, func() (contracts.Extractor[stateTestArtifact], error) { return stateTestExtractor{}, nil }))
|
||||
register(pipeline.RegisterMerger(registries.Mergers, pipeline.ModuleSpec{Key: "reference/shared-merge", Stage: pipeline.StageMerge, Requires: []string{"artifact"}, Provides: []string{"merged"}, ArtifactKind: referenceContractKindAlpha, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "alpha-merge"}, {Name: "required-merge", Required: true}}}, func() (contracts.Merger[stateTestArtifact], error) { return stateTestMerger{}, nil }))
|
||||
register(pipeline.RegisterExtractor(registries.Extractors, pipeline.ModuleSpec{Key: "reference/extract-alpha", Stage: pipeline.StageExtract, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: referenceContractKindAlpha, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "alpha-slot"}, {Name: "required-extract", Required: true}}}, func() (contracts.Extractor[stateTestArtifact], error) { return stateTestExtractor{}, nil }))
|
||||
register(pipeline.RegisterExtractor(registries.Extractors, pipeline.ModuleSpec{Key: "reference/extract-beta", Stage: pipeline.StageExtract, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: referenceContractKindBeta, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "beta-slot"}, {Name: "required-extract", Required: true}}}, func() (contracts.Extractor[stateTestArtifact], error) { return stateTestExtractor{}, nil }))
|
||||
register(pipeline.RegisterMerger(registries.Mergers, pipeline.ModuleSpec{Key: "reference/shared-merge", Stage: pipeline.StageMerge, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"artifact"}, Provides: []string{"merged"}, ArtifactKind: referenceContractKindAlpha, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "alpha-merge"}, {Name: "required-merge", Required: true}}}, func() (contracts.Merger[stateTestArtifact], error) { return stateTestMerger{}, nil }))
|
||||
if includeBetaMerger {
|
||||
register(pipeline.RegisterMerger(registries.Mergers, pipeline.ModuleSpec{Key: "reference/shared-merge", Stage: pipeline.StageMerge, Requires: []string{"artifact"}, Provides: []string{"merged"}, ArtifactKind: referenceContractKindBeta, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "beta-merge"}, {Name: "required-merge", Required: true}}}, func() (contracts.Merger[stateTestArtifact], error) { return stateTestMerger{}, nil }))
|
||||
register(pipeline.RegisterMerger(registries.Mergers, pipeline.ModuleSpec{Key: "reference/shared-merge", Stage: pipeline.StageMerge, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"artifact"}, Provides: []string{"merged"}, ArtifactKind: referenceContractKindBeta, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "beta-merge"}, {Name: "required-merge", Required: true}}}, func() (contracts.Merger[stateTestArtifact], error) { return stateTestMerger{}, nil }))
|
||||
}
|
||||
register(pipeline.RegisterNormalizer(registries.Normalizers, pipeline.ModuleSpec{Key: "reference/shared-normalize", Stage: pipeline.StageNormalize, Requires: []string{"merged"}, Provides: []string{"normalized"}, ArtifactKind: referenceContractKindAlpha, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "alpha-normalize"}, {Name: "required-normalize", Required: true}}}, func() (contracts.Normalizer[stateTestArtifact], error) { return stateTestNormalizer{}, nil }))
|
||||
register(pipeline.RegisterNormalizer(registries.Normalizers, pipeline.ModuleSpec{Key: "reference/shared-normalize", Stage: pipeline.StageNormalize, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"merged"}, Provides: []string{"normalized"}, ArtifactKind: referenceContractKindAlpha, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "alpha-normalize"}, {Name: "required-normalize", Required: true}}}, func() (contracts.Normalizer[stateTestArtifact], error) { return stateTestNormalizer{}, nil }))
|
||||
if includeBetaNormalizer {
|
||||
register(pipeline.RegisterNormalizer(registries.Normalizers, pipeline.ModuleSpec{Key: "reference/shared-normalize", Stage: pipeline.StageNormalize, Requires: []string{"merged"}, Provides: []string{"normalized"}, ArtifactKind: referenceContractKindBeta, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "beta-normalize"}, {Name: "required-normalize", Required: true}}}, func() (contracts.Normalizer[stateTestArtifact], error) { return stateTestNormalizer{}, nil }))
|
||||
register(pipeline.RegisterNormalizer(registries.Normalizers, pipeline.ModuleSpec{Key: "reference/shared-normalize", Stage: pipeline.StageNormalize, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"merged"}, Provides: []string{"normalized"}, ArtifactKind: referenceContractKindBeta, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "beta-normalize"}, {Name: "required-normalize", Required: true}}}, func() (contracts.Normalizer[stateTestArtifact], error) { return stateTestNormalizer{}, nil }))
|
||||
}
|
||||
register(registries.Outputs.RegisterWithSpec(pipeline.ModuleSpec{Key: "reference/output", Stage: pipeline.StageOutput, Requires: []string{"normalized"}, Provides: []string{"output"}}, func() (contracts.OutputEncoder, error) { return stateTestOutput{}, nil }))
|
||||
register(registries.Outputs.RegisterWithSpec(pipeline.ModuleSpec{Key: "reference/output", Stage: pipeline.StageOutput, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"normalized"}, Provides: []string{"output"}}, func() (contracts.OutputEncoder, error) { return stateTestOutput{}, nil }))
|
||||
return catalogFromRegistries(registries)
|
||||
}
|
||||
|
||||
@@ -418,11 +418,13 @@ func (referenceContractCodecB) Decode([]byte) (stateTestArtifact, error) {
|
||||
|
||||
func referenceContractLane(t *testing.T, resolved pipeline.ResolvedPipeline, id string) pipeline.ResolvedArtifactLane {
|
||||
t.Helper()
|
||||
for _, lane := range resolved.Steps[0].ArtifactLanes {
|
||||
for _, step := range resolved.Steps {
|
||||
for _, lane := range step.ArtifactLanes {
|
||||
if lane.ID == id {
|
||||
return lane
|
||||
}
|
||||
}
|
||||
}
|
||||
t.Fatalf("lane %q not found", id)
|
||||
return pipeline.ResolvedArtifactLane{}
|
||||
}
|
||||
|
||||
@@ -5,6 +5,7 @@ import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
@@ -22,13 +23,14 @@ import (
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/chunkplan"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
frameworkdebug "gitea.maximumdirect.net/eric/notarius/internal/framework/debug"
|
||||
frameworkllm "gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
)
|
||||
|
||||
const defaultConfigPath = "/usr/local/etc/notarius/config.yml"
|
||||
const usage = `Usage:
|
||||
notarius help
|
||||
notarius run <pipeline-id> --input path/to/source.json [--config path/to/config.yml] [--output-dir path] [--chunk_cache auto|bypass|refresh] [--resume] [--recompute-step step-id] [--debug [--debug-dir path]] [--only lane-a,lane-b] [--session-id id] [--reference selector=path] [--without-reference selector]
|
||||
notarius run <pipeline-id> --input path/to/source.json [--json] [flags]
|
||||
notarius config validate --config path/to/config.yml [--pipeline pipeline-id] [--only lane-a,lane-b]
|
||||
notarius pipelines list --config path/to/config.yml [--json]
|
||||
`
|
||||
@@ -44,9 +46,14 @@ type Options struct {
|
||||
ChunkPlanStoreFactory pipeline.ChunkPlanStoreFactory
|
||||
DebugRecorderFactory func(string) (pipeline.DebugRecorder, error)
|
||||
DebugTerminalFactory func(*debugbundle.SummaryWriter) DebugTerminalWriter
|
||||
promptKitAssets *frameworkllm.AssetRegistry
|
||||
}
|
||||
|
||||
type LLMClientFactory func(ctx context.Context, cfg config.Config, profileID string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error)
|
||||
type LLMRuntimeOverrides struct {
|
||||
ReasoningEffort *string
|
||||
}
|
||||
|
||||
type LLMClientFactory func(ctx context.Context, cfg config.Config, profileID string, overrides LLMRuntimeOverrides) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error)
|
||||
|
||||
// Run executes the command-line interface and returns a process exit code.
|
||||
func Run(args []string, stdout, stderr io.Writer) int {
|
||||
@@ -115,12 +122,17 @@ func normalizeOptions(opts Options) (Options, error) {
|
||||
}
|
||||
opts.Registries = components.registries
|
||||
opts.Catalog = catalogFromRegistries(components.registries)
|
||||
if opts.LLMClientFactory == nil {
|
||||
opts.LLMClientFactory = productionLLMClientFactoryWithAssets(components.assets)
|
||||
}
|
||||
opts.promptKitAssets = components.assets
|
||||
}
|
||||
if opts.LLMClientFactory == nil {
|
||||
opts.LLMClientFactory = productionLLMClientFactory
|
||||
if opts.promptKitAssets == nil {
|
||||
assets, err := productionPromptAssets()
|
||||
if err != nil {
|
||||
return Options{}, err
|
||||
}
|
||||
opts.promptKitAssets = assets
|
||||
}
|
||||
opts.LLMClientFactory = productionLLMClientFactoryWithAssets(opts.promptKitAssets)
|
||||
}
|
||||
return opts, nil
|
||||
}
|
||||
@@ -132,16 +144,20 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
|
||||
inputPath := fs.String("input", "", "source input file path")
|
||||
onlyRaw := fs.String("only", "", "comma-separated artifact lanes")
|
||||
outputDir := fs.String("output-dir", "", "output directory")
|
||||
machineOutput := fs.Bool("json", false, "write the successful run result as JSON")
|
||||
debug := fs.Bool("debug", false, "write a debug bundle")
|
||||
debugDir := fs.String("debug-dir", "", "debug bundle directory")
|
||||
llmProfile := fs.String("llm-profile", "", "LLM profile override")
|
||||
reasoningEffort := singleValueFlag{name: "--reasoning-effort"}
|
||||
clearReasoningEffort := fs.Bool("clear-reasoning-effort", false, "clear the LLM profile reasoning effort")
|
||||
resume := fs.Bool("resume", false, "reuse compatible recorded checkpoints")
|
||||
recomputeStep := singleValueFlag{}
|
||||
recomputeStep := singleValueFlag{name: "--recompute-step"}
|
||||
chunkCache := chunkCacheFlag{}
|
||||
sessionID := sessionIDFlag{}
|
||||
referenceFlags := stringListFlag{}
|
||||
withoutReferenceFlags := stringListFlag{}
|
||||
fs.Var(&sessionID, "session-id", "prompt session identifier")
|
||||
fs.Var(&reasoningEffort, "reasoning-effort", "reasoning effort override")
|
||||
fs.Var(&chunkCache, "chunk_cache", "chunk plan cache mode: auto, bypass, or refresh")
|
||||
fs.Var(&referenceFlags, "reference", "reference binding, as slot=path, chunk.slot=path, merge.slot=path, lane.slot=path, lane.extract.slot=path, lane.merge.slot=path, or lane.normalize.slot=path")
|
||||
fs.Var(&withoutReferenceFlags, "without-reference", "unbind a reference, using the same selector forms as --reference")
|
||||
@@ -187,6 +203,22 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
|
||||
fmt.Fprintln(stderr, "notarius: --session-id must not be empty")
|
||||
return 2
|
||||
}
|
||||
if reasoningEffort.set && *clearReasoningEffort {
|
||||
fmt.Fprintln(stderr, "notarius: --reasoning-effort cannot be combined with --clear-reasoning-effort")
|
||||
return 2
|
||||
}
|
||||
if reasoningEffort.set && strings.TrimSpace(reasoningEffort.value) == "" {
|
||||
fmt.Fprintln(stderr, "notarius: --reasoning-effort must not be empty")
|
||||
return 2
|
||||
}
|
||||
runtimeOverrides := LLMRuntimeOverrides{}
|
||||
if reasoningEffort.set {
|
||||
value := strings.TrimSpace(reasoningEffort.value)
|
||||
runtimeOverrides.ReasoningEffort = &value
|
||||
} else if *clearReasoningEffort {
|
||||
value := ""
|
||||
runtimeOverrides.ReasoningEffort = &value
|
||||
}
|
||||
only, err := parseOnly(*onlyRaw)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "notarius: %v\n", err)
|
||||
@@ -284,6 +316,7 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
|
||||
ConfigSource: configSource(*configPath),
|
||||
OnlyLanes: append([]string(nil), only...),
|
||||
ChunkCacheOverride: chunkCache.explicitValue(),
|
||||
ReasoningEffortOverride: runtimeOverrides.ReasoningEffort,
|
||||
Resume: *resume,
|
||||
RecomputeStep: strings.TrimSpace(recomputeStep.value),
|
||||
RunID: runID,
|
||||
@@ -313,7 +346,7 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
|
||||
return failPipelineCommand(stderr, commandState, terminalWriter, err)
|
||||
}
|
||||
profileIDs := effectiveLLMProfileIDs(effective.ResolvedPipeline)
|
||||
if err := validateExplicitScriptoriumProfiles(context.Background(), effective.Config, profileIDs); err != nil {
|
||||
if err := validateExplicitPromptKitProfiles(context.Background(), effective.Config, profileIDs, opts.promptKitAssets); err != nil {
|
||||
return failPipelineCommand(stderr, commandState, terminalWriter, err)
|
||||
}
|
||||
workingDir, err := os.Getwd()
|
||||
@@ -361,10 +394,17 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
|
||||
if len(profileIDs) == 1 {
|
||||
factoryProfileID = profileIDs[0]
|
||||
}
|
||||
llmClient, llmProfiles, err := opts.LLMClientFactory(ctx, effective.Config, factoryProfileID)
|
||||
llmClient, llmProfiles, err := opts.LLMClientFactory(ctx, effective.Config, factoryProfileID, runtimeOverrides)
|
||||
if err != nil {
|
||||
return failPipelineCommand(stderr, commandState, terminalWriter, fmt.Errorf("create LLM client for profile %q: %w", factoryProfileID, err))
|
||||
}
|
||||
var llmFingerprints []checkpoint.Fingerprint
|
||||
if effective.Config.Cache.Checkpoints.Enabled {
|
||||
llmFingerprints, err = llmCheckpointFingerprints(llmClient)
|
||||
if err != nil {
|
||||
return failPipelineCommand(stderr, commandState, terminalWriter, fmt.Errorf("prepare LLM checkpoint identity: %w", err))
|
||||
}
|
||||
}
|
||||
llmClient = pipeline.WithDebugLLMRecording(llmClient, debugRecorder)
|
||||
prepared, err := pipeline.Prepare(effective.ResolvedPipeline, registries, pipeline.ModuleDependencies{LLM: llmClient})
|
||||
if err != nil {
|
||||
@@ -378,7 +418,7 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
|
||||
if err != nil {
|
||||
return failPipelineCommand(stderr, commandState, terminalWriter, err)
|
||||
}
|
||||
checkpointRecorder, checkpointLoader, err := checkpointHandlersForRun(effective.Config.Cache.Checkpoints, opts, effective.ResolvedPipeline, prepared.CheckpointFingerprints(), rawInput, only, llmProfiles, strings.TrimSpace(*llmProfile), strings.TrimSpace(sessionID.value), *resume)
|
||||
checkpointRecorder, checkpointLoader, err := checkpointHandlersForRun(effective.Config.Cache.Checkpoints, opts, effective.ResolvedPipeline, prepared.CheckpointFingerprints(), llmFingerprints, rawInput, only, llmProfiles, strings.TrimSpace(*llmProfile), strings.TrimSpace(sessionID.value), runtimeOverrides, *resume)
|
||||
if err != nil {
|
||||
return failPipelineCommand(stderr, commandState, terminalWriter, err)
|
||||
}
|
||||
@@ -415,6 +455,17 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
|
||||
if err := writePartialSummary(summary, output); err != nil {
|
||||
return failPipelineCommand(stderr, commandState, terminalWriter, fmt.Errorf("write debug summary: %w", err))
|
||||
}
|
||||
var encodedResult []byte
|
||||
if *machineOutput {
|
||||
result, err := newRunResult(effective.ResolvedPipeline, output, runOutputDir, debugPath)
|
||||
if err != nil {
|
||||
return failPipelineCommand(stderr, commandState, terminalWriter, err)
|
||||
}
|
||||
encodedResult, err = encodeRunResult(result)
|
||||
if err != nil {
|
||||
return failPipelineCommand(stderr, commandState, terminalWriter, err)
|
||||
}
|
||||
}
|
||||
if err := writeOutputFiles(runOutputDir, output.OutputFiles); err != nil {
|
||||
return failPipelineCommand(stderr, commandState, terminalWriter, err)
|
||||
}
|
||||
@@ -422,10 +473,16 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
|
||||
return writePipelineCommandFailure(stderr, commandState, primaryErr, persistenceErr)
|
||||
}
|
||||
|
||||
if *machineOutput {
|
||||
if err := writeRunResult(stdout, encodedResult); err != nil {
|
||||
return writePipelineCommandFailure(stderr, commandState, errors.New("write run result"), nil)
|
||||
}
|
||||
} else {
|
||||
fmt.Fprintf(stdout, "pipeline %q complete: outputs=%d rejected=%d output=%s\n", effective.PipelineID, len(output.NormalizeOutputs), len(output.Rejected), runOutputDir)
|
||||
if debugPath != "" {
|
||||
fmt.Fprintf(stdout, "debug=%s\n", debugPath)
|
||||
}
|
||||
}
|
||||
if len(output.Warnings) > 0 {
|
||||
fmt.Fprintf(stderr, "notarius: run completed with %d warning(s)\n", len(output.Warnings))
|
||||
}
|
||||
@@ -462,11 +519,13 @@ func checkpointHandlersForRun(
|
||||
opts Options,
|
||||
resolved pipeline.ResolvedPipeline,
|
||||
componentFingerprints []pipeline.CheckpointFingerprint,
|
||||
llmFingerprints []checkpoint.Fingerprint,
|
||||
rawInput []byte,
|
||||
only []string,
|
||||
llmProfiles []artifacts.LLMProfileManifest,
|
||||
llmProfileOverride string,
|
||||
sessionID string,
|
||||
runtimeOverrides LLMRuntimeOverrides,
|
||||
resume bool,
|
||||
) (pipeline.CheckpointRecorder, pipeline.CheckpointLoader, error) {
|
||||
if !settings.Enabled {
|
||||
@@ -480,9 +539,13 @@ func checkpointHandlersForRun(
|
||||
InputKey: resolved.Input.Module,
|
||||
RawInputDigest: rawInputDigest(rawInput),
|
||||
SelectedLanes: only,
|
||||
RuntimeOverrides: runtimeOverrideFingerprints(llmProfileOverride, sessionID),
|
||||
RuntimeOverrides: runtimeOverrideFingerprints(llmProfileOverride, sessionID, runtimeOverrides),
|
||||
References: pipeline.ReferenceProvenance(resolved),
|
||||
ProvenanceFingerprints: append(llmProfileFingerprints(llmProfiles), checkpointIdentityFingerprints(componentFingerprints)...),
|
||||
ProvenanceFingerprints: combineCheckpointFingerprints(
|
||||
llmProfileFingerprints(llmProfiles),
|
||||
llmFingerprints,
|
||||
checkpointIdentityFingerprints(componentFingerprints),
|
||||
),
|
||||
})
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("create checkpoint identity: %w", err)
|
||||
@@ -508,6 +571,30 @@ func checkpointHandlersForRun(
|
||||
return recorder, loader, nil
|
||||
}
|
||||
|
||||
func llmCheckpointFingerprints(client contracts.StructuredLLMClient) ([]checkpoint.Fingerprint, error) {
|
||||
provider, ok := client.(frameworkllm.CheckpointFingerprintProvider)
|
||||
if !ok {
|
||||
return nil, nil
|
||||
}
|
||||
values, err := provider.LLMCheckpointFingerprints()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := make([]checkpoint.Fingerprint, 0, len(values))
|
||||
for _, value := range values {
|
||||
out = append(out, checkpoint.Fingerprint{Name: value.Name, Value: value.Value})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func combineCheckpointFingerprints(sources ...[]checkpoint.Fingerprint) []checkpoint.Fingerprint {
|
||||
var out []checkpoint.Fingerprint
|
||||
for _, source := range sources {
|
||||
out = append(out, source...)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func recomputePolicy(resolved pipeline.ResolvedPipeline, requestedStep string) (pipeline.CheckpointExecutionPolicy, error) {
|
||||
requestedStep = strings.TrimSpace(requestedStep)
|
||||
if requestedStep == "" {
|
||||
@@ -613,7 +700,7 @@ func rawInputDigest(data []byte) string {
|
||||
return "sha256:" + hex.EncodeToString(sum[:])
|
||||
}
|
||||
|
||||
func runtimeOverrideFingerprints(llmProfileOverride string, sessionID string) []checkpoint.Fingerprint {
|
||||
func runtimeOverrideFingerprints(llmProfileOverride string, sessionID string, runtimeOverrides LLMRuntimeOverrides) []checkpoint.Fingerprint {
|
||||
var values []checkpoint.Fingerprint
|
||||
if strings.TrimSpace(llmProfileOverride) != "" {
|
||||
values = append(values, checkpoint.Fingerprint{Name: "llm_profile_override", Value: strings.TrimSpace(llmProfileOverride)})
|
||||
@@ -621,6 +708,13 @@ func runtimeOverrideFingerprints(llmProfileOverride string, sessionID string) []
|
||||
if strings.TrimSpace(sessionID) != "" {
|
||||
values = append(values, checkpoint.Fingerprint{Name: "session_id", Value: strings.TrimSpace(sessionID)})
|
||||
}
|
||||
if runtimeOverrides.ReasoningEffort != nil {
|
||||
value := strings.TrimSpace(*runtimeOverrides.ReasoningEffort)
|
||||
if value == "" {
|
||||
value = "<cleared>"
|
||||
}
|
||||
values = append(values, checkpoint.Fingerprint{Name: "reasoning_effort_override", Value: value})
|
||||
}
|
||||
return values
|
||||
}
|
||||
|
||||
@@ -777,7 +871,7 @@ func reorderRunArgs(args []string) []string {
|
||||
|
||||
func runFlagTakesValue(arg string) bool {
|
||||
switch arg {
|
||||
case "--config", "--input", "--only", "--output-dir", "--debug-dir", "--llm-profile", "--session-id", "--chunk_cache", "--reference", "--without-reference", "--recompute-step":
|
||||
case "--config", "--input", "--only", "--output-dir", "--debug-dir", "--llm-profile", "--session-id", "--reasoning-effort", "--chunk_cache", "--reference", "--without-reference", "--recompute-step":
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
@@ -837,11 +931,11 @@ func chunkPlanStoreForRun(cfg config.ChunkPlanCacheConfig, opts Options) (pipeli
|
||||
|
||||
func validateRunFlagValues(args []string) error {
|
||||
for i, arg := range args {
|
||||
if arg != "--session-id" {
|
||||
if arg != "--session-id" && arg != "--reasoning-effort" {
|
||||
continue
|
||||
}
|
||||
if i+1 >= len(args) || strings.HasPrefix(args[i+1], "-") {
|
||||
return fmt.Errorf("flag needs an argument: --session-id")
|
||||
return fmt.Errorf("flag needs an argument: %s", arg)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
@@ -864,12 +958,23 @@ func effectiveLLMProfileIDs(resolved pipeline.ResolvedPipeline) []string {
|
||||
seen[id] = struct{}{}
|
||||
}
|
||||
}
|
||||
if resolved.InputExecutionClass == contracts.ExecutionClassLLMBacked {
|
||||
add(resolved.Input)
|
||||
}
|
||||
if resolved.ChunkExecutionClass == contracts.ExecutionClassLLMBacked {
|
||||
add(resolved.Chunk)
|
||||
}
|
||||
for _, lane := range resolved.AllArtifactLanes() {
|
||||
if lane.ExtractExecutionClass == contracts.ExecutionClassLLMBacked {
|
||||
add(lane.Extract)
|
||||
}
|
||||
if lane.MergeExecutionClass == contracts.ExecutionClassLLMBacked {
|
||||
add(lane.Merge)
|
||||
}
|
||||
if lane.NormalizeExecutionClass == contracts.ExecutionClassLLMBacked {
|
||||
add(lane.Normalize)
|
||||
}
|
||||
}
|
||||
for _, chain := range resolved.ValidatorChains {
|
||||
for _, validator := range chain.Validators {
|
||||
if validator.ExecutionClass == contracts.ExecutionClassLLMBacked {
|
||||
@@ -877,6 +982,9 @@ func effectiveLLMProfileIDs(resolved pipeline.ResolvedPipeline) []string {
|
||||
}
|
||||
}
|
||||
}
|
||||
if resolved.OutputExecutionClass == contracts.ExecutionClassLLMBacked {
|
||||
add(resolved.Output)
|
||||
}
|
||||
ids := make([]string, 0, len(seen))
|
||||
for id := range seen {
|
||||
ids = append(ids, id)
|
||||
@@ -960,7 +1068,7 @@ func runConfigValidate(args []string, stdout, stderr io.Writer, opts Options) in
|
||||
fmt.Fprintf(stderr, "notarius: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
if err := validateExplicitScriptoriumProfiles(context.Background(), effective.Config, effectiveLLMProfileIDs(effective.ResolvedPipeline)); err != nil {
|
||||
if err := validateExplicitPromptKitProfiles(context.Background(), effective.Config, effectiveLLMProfileIDs(effective.ResolvedPipeline), opts.promptKitAssets); err != nil {
|
||||
fmt.Fprintf(stderr, "notarius: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
@@ -1125,6 +1233,7 @@ type sessionIDFlag struct {
|
||||
}
|
||||
|
||||
type singleValueFlag struct {
|
||||
name string
|
||||
value string
|
||||
set bool
|
||||
}
|
||||
@@ -1138,7 +1247,7 @@ func (flag *singleValueFlag) String() string {
|
||||
|
||||
func (flag *singleValueFlag) Set(value string) error {
|
||||
if flag.set {
|
||||
return fmt.Errorf("--recompute-step may be specified only once")
|
||||
return fmt.Errorf("%s may be specified only once", flag.name)
|
||||
}
|
||||
flag.value = value
|
||||
flag.set = true
|
||||
|
||||
@@ -12,6 +12,7 @@ import (
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/checkpoint"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
)
|
||||
@@ -241,12 +242,14 @@ func TestRunLLMProfileOverrideAndValidationUseInjectedBoundaries(t *testing.T) {
|
||||
t.Run("one effective profile reaches the factory and modules", func(t *testing.T) {
|
||||
roots := newStateTestRoots(t)
|
||||
profileDir := writeRunContractProfiles(t, "override-profile")
|
||||
prependRunContractConfig(t, roots, fmt.Sprintf("scriptorium:\n profile_dir: %q\n", profileDir))
|
||||
prependRunContractConfig(t, roots, fmt.Sprintf("promptkit:\n profile_dir: %q\n", profileDir))
|
||||
harness := newStateTestHarness()
|
||||
var factoryProfiles []string
|
||||
opts := harness.options()
|
||||
opts.LLMClientFactory = func(_ context.Context, _ config.Config, profileID string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
var factoryOverrides []LLMRuntimeOverrides
|
||||
opts.LLMClientFactory = func(_ context.Context, _ config.Config, profileID string, overrides LLMRuntimeOverrides) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
factoryProfiles = append(factoryProfiles, profileID)
|
||||
factoryOverrides = append(factoryOverrides, overrides)
|
||||
return nil, nil, nil
|
||||
}
|
||||
var stdout, stderr bytes.Buffer
|
||||
@@ -257,6 +260,9 @@ func TestRunLLMProfileOverrideAndValidationUseInjectedBoundaries(t *testing.T) {
|
||||
if len(factoryProfiles) != 1 || factoryProfiles[0] != "override-profile" {
|
||||
t.Fatalf("factory profiles = %#v, want one override profile", factoryProfiles)
|
||||
}
|
||||
if len(factoryOverrides) != 1 || factoryOverrides[0].ReasoningEffort != nil {
|
||||
t.Fatalf("factory overrides = %#v, want inherited reasoning", factoryOverrides)
|
||||
}
|
||||
harness.mu.Lock()
|
||||
profiles := append([]string(nil), harness.moduleProfiles...)
|
||||
harness.mu.Unlock()
|
||||
@@ -270,16 +276,16 @@ func TestRunLLMProfileOverrideAndValidationUseInjectedBoundaries(t *testing.T) {
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("validator profile remains distinct", func(t *testing.T) {
|
||||
t.Run("runtime override applies to validators", func(t *testing.T) {
|
||||
roots := newStateTestRoots(t)
|
||||
profileDir := writeRunContractProfiles(t, "override-profile", "validator-profile")
|
||||
prependRunContractConfig(t, roots, fmt.Sprintf("scriptorium:\n profile_dir: %q\n", profileDir))
|
||||
prependRunContractConfig(t, roots, fmt.Sprintf("promptkit:\n profile_dir: %q\n", profileDir))
|
||||
harness := newStateTestHarness()
|
||||
var validatorProfiles []string
|
||||
opts := harness.options()
|
||||
registerRunContractValidator(t, &opts, &validatorProfiles)
|
||||
factoryProfiles := []string{}
|
||||
opts.LLMClientFactory = func(_ context.Context, _ config.Config, profileID string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
opts.LLMClientFactory = func(_ context.Context, _ config.Config, profileID string, _ LLMRuntimeOverrides) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
factoryProfiles = append(factoryProfiles, profileID)
|
||||
return nil, nil, nil
|
||||
}
|
||||
@@ -288,21 +294,21 @@ func TestRunLLMProfileOverrideAndValidationUseInjectedBoundaries(t *testing.T) {
|
||||
if code != 0 || stderr.Len() != 0 {
|
||||
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||
}
|
||||
if len(factoryProfiles) != 1 || factoryProfiles[0] != "" {
|
||||
t.Fatalf("factory profiles = %#v, want one call without a unique profile", factoryProfiles)
|
||||
if len(factoryProfiles) != 1 || factoryProfiles[0] != "override-profile" {
|
||||
t.Fatalf("factory profiles = %#v, want one override profile", factoryProfiles)
|
||||
}
|
||||
if len(validatorProfiles) != 1 || validatorProfiles[0] != "validator-profile" {
|
||||
t.Fatalf("validator profiles = %#v, want configured validator profile", validatorProfiles)
|
||||
if len(validatorProfiles) != 1 || validatorProfiles[0] != "override-profile" {
|
||||
t.Fatalf("validator profiles = %#v, want runtime override", validatorProfiles)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("unknown profile is rejected without factory access", func(t *testing.T) {
|
||||
roots := newStateTestRoots(t)
|
||||
profileDir := writeRunContractProfiles(t, "override-profile")
|
||||
prependRunContractConfig(t, roots, fmt.Sprintf("scriptorium:\n profile_dir: %q\n", profileDir))
|
||||
prependRunContractConfig(t, roots, fmt.Sprintf("promptkit:\n profile_dir: %q\n", profileDir))
|
||||
factoryCalls := 0
|
||||
opts := newStateTestHarness().options()
|
||||
opts.LLMClientFactory = func(context.Context, config.Config, string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
opts.LLMClientFactory = func(context.Context, config.Config, string, LLMRuntimeOverrides) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
factoryCalls++
|
||||
return nil, nil, nil
|
||||
}
|
||||
@@ -312,18 +318,164 @@ func TestRunLLMProfileOverrideAndValidationUseInjectedBoundaries(t *testing.T) {
|
||||
t.Fatalf("code=%d stdout=%q stderr=%q factoryCalls=%d", code, stdout.String(), stderr.String(), factoryCalls)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("pipeline default is rejected before factory access", func(t *testing.T) {
|
||||
roots := newStateTestRoots(t)
|
||||
profileDir := writeRunContractProfiles(t, "configured-profile")
|
||||
prependRunContractConfig(t, roots, fmt.Sprintf("promptkit:\n profile_dir: %q\n", profileDir))
|
||||
replaceStateTestConfigLine(t, roots.config, " sample:\n", " sample:\n llm_profile: missing-profile\n")
|
||||
factoryCalls := 0
|
||||
opts := newStateTestHarness().options()
|
||||
opts.LLMClientFactory = func(context.Context, config.Config, string, LLMRuntimeOverrides) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
factoryCalls++
|
||||
return nil, nil, nil
|
||||
}
|
||||
var stdout, stderr bytes.Buffer
|
||||
code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass"}, &stdout, &stderr, opts)
|
||||
if code != 1 || !strings.Contains(stderr.String(), "not configured") || factoryCalls != 0 || stdout.Len() != 0 {
|
||||
t.Fatalf("code=%d stdout=%q stderr=%q factoryCalls=%d", code, stdout.String(), stderr.String(), factoryCalls)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestRunReasoningEffortOverrideReachesFactory(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
flags []string
|
||||
wantValue string
|
||||
wantSet bool
|
||||
}{
|
||||
{name: "inherit"},
|
||||
{name: "replace", flags: []string{"--reasoning-effort", " focused "}, wantValue: "focused", wantSet: true},
|
||||
{name: "clear", flags: []string{"--clear-reasoning-effort"}, wantSet: true},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
roots := newStateTestRoots(t)
|
||||
opts := newStateTestHarness().options()
|
||||
var got []LLMRuntimeOverrides
|
||||
opts.LLMClientFactory = func(_ context.Context, _ config.Config, _ string, overrides LLMRuntimeOverrides) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
got = append(got, overrides)
|
||||
return nil, nil, nil
|
||||
}
|
||||
args := append([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass"}, tt.flags...)
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := RunWithOptions(args, &stdout, &stderr, opts); code != 0 || stderr.Len() != 0 {
|
||||
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||
}
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("factory overrides = %#v, want one call", got)
|
||||
}
|
||||
if !tt.wantSet {
|
||||
if got[0].ReasoningEffort != nil {
|
||||
t.Fatalf("reasoning effort = %q, want inherit", *got[0].ReasoningEffort)
|
||||
}
|
||||
return
|
||||
}
|
||||
if got[0].ReasoningEffort == nil || *got[0].ReasoningEffort != tt.wantValue {
|
||||
t.Fatalf("reasoning effort = %#v, want %q", got[0].ReasoningEffort, tt.wantValue)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunReasoningEffortOverrideRejectsInvalidSyntax(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
flags []string
|
||||
wantError string
|
||||
}{
|
||||
{
|
||||
name: "mutually exclusive controls",
|
||||
flags: []string{"--reasoning-effort", "focused", "--clear-reasoning-effort"},
|
||||
wantError: "cannot be combined",
|
||||
},
|
||||
{
|
||||
name: "empty replacement",
|
||||
flags: []string{"--reasoning-effort", " "},
|
||||
wantError: "must not be empty",
|
||||
},
|
||||
{
|
||||
name: "duplicate replacement",
|
||||
flags: []string{"--reasoning-effort", "low", "--reasoning-effort", "high"},
|
||||
wantError: "may be specified only once",
|
||||
},
|
||||
{
|
||||
name: "missing replacement",
|
||||
flags: []string{"--reasoning-effort"},
|
||||
wantError: "flag needs an argument",
|
||||
},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
roots := newStateTestRoots(t)
|
||||
args := append([]string{"run", "sample", "--config", roots.config, "--input", roots.input}, tt.flags...)
|
||||
var stdout, stderr bytes.Buffer
|
||||
code := RunWithOptions(args, &stdout, &stderr, newStateTestHarness().options())
|
||||
if code != 2 || stdout.Len() != 0 || !strings.Contains(stderr.String(), tt.wantError) {
|
||||
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||
}
|
||||
assertNoRunState(t, roots)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestReasoningEffortOverrideSeparatesCheckpointIdentities(t *testing.T) {
|
||||
replacement := " focused "
|
||||
cleared := ""
|
||||
states := []struct {
|
||||
name string
|
||||
overrides LLMRuntimeOverrides
|
||||
wantValue string
|
||||
wantSet bool
|
||||
}{
|
||||
{name: "inherit"},
|
||||
{name: "replace", overrides: LLMRuntimeOverrides{ReasoningEffort: &replacement}, wantValue: "focused", wantSet: true},
|
||||
{name: "clear", overrides: LLMRuntimeOverrides{ReasoningEffort: &cleared}, wantValue: "<cleared>", wantSet: true},
|
||||
}
|
||||
digests := make(map[string]string, len(states))
|
||||
for _, state := range states {
|
||||
fingerprints := runtimeOverrideFingerprints("", "", state.overrides)
|
||||
var value string
|
||||
var found bool
|
||||
for _, fingerprint := range fingerprints {
|
||||
if fingerprint.Name == "reasoning_effort_override" {
|
||||
value, found = fingerprint.Value, true
|
||||
}
|
||||
}
|
||||
if found != state.wantSet || (found && value != state.wantValue) {
|
||||
t.Fatalf("%s fingerprint found=%t value=%q, want found=%t value=%q", state.name, found, value, state.wantSet, state.wantValue)
|
||||
}
|
||||
identity, err := checkpoint.NewIdentity(checkpoint.IdentityInput{
|
||||
Pipeline: pipeline.ResolvedPipeline{ID: "sample", Digest: "sha256:pipeline", Input: pipeline.Binding("test/input")},
|
||||
RawInputDigest: "sha256:input",
|
||||
RuntimeOverrides: fingerprints,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
digests[state.name] = identity.Digest
|
||||
}
|
||||
if digests["inherit"] == digests["replace"] || digests["inherit"] == digests["clear"] || digests["replace"] == digests["clear"] {
|
||||
t.Fatalf("checkpoint identity digests are not distinct: %#v", digests)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEffectiveLLMProfileIDsAreSortedDeduplicatedAndLLMOnly(t *testing.T) {
|
||||
resolved := pipeline.ResolvedPipeline{
|
||||
Input: pipeline.ModuleBinding{LLMProfile: "input-profile"},
|
||||
InputExecutionClass: contracts.ExecutionClassLLMBacked,
|
||||
Chunk: pipeline.ModuleBinding{LLMProfile: " zeta "},
|
||||
ChunkExecutionClass: contracts.ExecutionClassLLMBacked,
|
||||
Steps: []pipeline.ResolvedPipelineStep{{
|
||||
ID: "default",
|
||||
ArtifactLanes: []pipeline.ResolvedArtifactLane{{
|
||||
Extract: pipeline.ModuleBinding{LLMProfile: "alpha"},
|
||||
Merge: pipeline.ModuleBinding{LLMProfile: "zeta"},
|
||||
ExtractExecutionClass: contracts.ExecutionClassLLMBacked,
|
||||
Merge: pipeline.ModuleBinding{LLMProfile: "deterministic-merge"},
|
||||
MergeExecutionClass: contracts.ExecutionClassDeterministic,
|
||||
Normalize: pipeline.ModuleBinding{LLMProfile: " gamma "},
|
||||
NormalizeExecutionClass: contracts.ExecutionClassLLMBacked,
|
||||
}},
|
||||
}},
|
||||
ValidatorChains: []pipeline.ResolvedValidatorChain{{Validators: []pipeline.ResolvedValidator{
|
||||
@@ -331,9 +483,10 @@ func TestEffectiveLLMProfileIDsAreSortedDeduplicatedAndLLMOnly(t *testing.T) {
|
||||
{Binding: pipeline.ModuleBinding{LLMProfile: "beta"}, ExecutionClass: contracts.ExecutionClassLLMBacked},
|
||||
}}},
|
||||
Output: pipeline.ModuleBinding{LLMProfile: "output-profile"},
|
||||
OutputExecutionClass: contracts.ExecutionClassLLMBacked,
|
||||
}
|
||||
got := effectiveLLMProfileIDs(resolved)
|
||||
want := []string{"alpha", "beta", "gamma", "zeta"}
|
||||
want := []string{"alpha", "beta", "gamma", "input-profile", "output-profile", "zeta"}
|
||||
if strings.Join(got, ",") != strings.Join(want, ",") {
|
||||
t.Fatalf("effective profiles = %#v, want %#v", got, want)
|
||||
}
|
||||
@@ -376,7 +529,7 @@ func TestRunFactoryAndPreparationFailuresAreProcessFailures(t *testing.T) {
|
||||
t.Run("LLM factory", func(t *testing.T) {
|
||||
roots := newStateTestRoots(t)
|
||||
opts := newStateTestHarness().options()
|
||||
opts.LLMClientFactory = func(context.Context, config.Config, string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
opts.LLMClientFactory = func(context.Context, config.Config, string, LLMRuntimeOverrides) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
return nil, nil, errors.New("injected LLM factory failure")
|
||||
}
|
||||
var stdout, stderr bytes.Buffer
|
||||
@@ -397,7 +550,7 @@ func TestRunFactoryAndPreparationFailuresAreProcessFailures(t *testing.T) {
|
||||
t.Fatal(err)
|
||||
}
|
||||
opts := newStateTestHarness().options()
|
||||
if err := pipeline.RegisterExtractorBuilder(opts.Registries.Extractors, pipeline.ModuleSpec{Key: "test/failing-extract", Stage: pipeline.StageExtract, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: stateTestArtifactKind}, func(map[string]any) error { return nil }, func(pipeline.BuildRequest) (contracts.Extractor[stateTestArtifact], error) {
|
||||
if err := pipeline.RegisterExtractorBuilder(opts.Registries.Extractors, pipeline.ModuleSpec{Key: "test/failing-extract", Stage: pipeline.StageExtract, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: stateTestArtifactKind}, func(map[string]any) error { return nil }, func(pipeline.BuildRequest) (contracts.Extractor[stateTestArtifact], error) {
|
||||
return nil, errors.New("injected extractor construction failure")
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
|
||||
111
internal/cli/run_result.go
Normal file
111
internal/cli/run_result.go
Normal file
@@ -0,0 +1,111 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
)
|
||||
|
||||
const runResultSchemaVersion = "notarius.run-result.v1"
|
||||
|
||||
type runResult struct {
|
||||
SchemaVersion string `json:"schema_version"`
|
||||
RunID string `json:"run_id"`
|
||||
PipelineID string `json:"pipeline_id"`
|
||||
OutputDirectory string `json:"output_directory"`
|
||||
IndexFile string `json:"index_file,omitempty"`
|
||||
NormalizedOutputCount int `json:"normalized_output_count"`
|
||||
RejectedOutputCount int `json:"rejected_output_count"`
|
||||
WarningCount int `json:"warning_count"`
|
||||
ValidationStatus string `json:"validation_status"`
|
||||
DebugDirectory string `json:"debug_directory,omitempty"`
|
||||
}
|
||||
|
||||
func newRunResult(resolved pipeline.ResolvedPipeline, output pipeline.RunOutput, outputDirectory, debugDirectory string) (runResult, error) {
|
||||
if strings.TrimSpace(output.Manifest.RunID) == "" {
|
||||
return runResult{}, fmt.Errorf("run result requires a run ID")
|
||||
}
|
||||
if strings.TrimSpace(resolved.ID) == "" {
|
||||
return runResult{}, fmt.Errorf("run result requires a resolved pipeline ID")
|
||||
}
|
||||
if strings.TrimSpace(output.Manifest.PipelineID) == "" {
|
||||
return runResult{}, fmt.Errorf("run result requires a manifest pipeline ID")
|
||||
}
|
||||
if output.Manifest.PipelineID != resolved.ID {
|
||||
return runResult{}, fmt.Errorf("run result pipeline ID does not match resolved pipeline")
|
||||
}
|
||||
if strings.TrimSpace(output.Manifest.ValidationStatus) == "" {
|
||||
return runResult{}, fmt.Errorf("run result requires a validation status")
|
||||
}
|
||||
if strings.TrimSpace(outputDirectory) == "" {
|
||||
return runResult{}, fmt.Errorf("run result requires an output directory")
|
||||
}
|
||||
|
||||
absOutputDirectory, err := filepath.Abs(outputDirectory)
|
||||
if err != nil {
|
||||
return runResult{}, fmt.Errorf("make output directory absolute: %w", err)
|
||||
}
|
||||
|
||||
result := runResult{
|
||||
SchemaVersion: runResultSchemaVersion,
|
||||
RunID: output.Manifest.RunID,
|
||||
PipelineID: output.Manifest.PipelineID,
|
||||
OutputDirectory: absOutputDirectory,
|
||||
NormalizedOutputCount: len(output.NormalizeOutputs),
|
||||
RejectedOutputCount: len(output.Rejected),
|
||||
WarningCount: len(output.Warnings),
|
||||
ValidationStatus: output.Manifest.ValidationStatus,
|
||||
}
|
||||
|
||||
if strings.TrimSpace(debugDirectory) != "" {
|
||||
absDebugDirectory, err := filepath.Abs(debugDirectory)
|
||||
if err != nil {
|
||||
return runResult{}, fmt.Errorf("make debug directory absolute: %w", err)
|
||||
}
|
||||
result.DebugDirectory = absDebugDirectory
|
||||
}
|
||||
|
||||
if resolved.Output.Module == pipeline.DefaultOutputModule {
|
||||
indexCount := 0
|
||||
for _, file := range output.OutputFiles {
|
||||
if file.Name == "index.json" {
|
||||
indexCount++
|
||||
}
|
||||
}
|
||||
if indexCount != 1 {
|
||||
return runResult{}, fmt.Errorf("production JSON output must contain exactly one index.json file")
|
||||
}
|
||||
result.IndexFile = "index.json"
|
||||
}
|
||||
|
||||
return result, nil
|
||||
}
|
||||
|
||||
func encodeRunResult(result runResult) ([]byte, error) {
|
||||
encoded, err := json.Marshal(result)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("encode run result: %w", err)
|
||||
}
|
||||
return append(encoded, '\n'), nil
|
||||
}
|
||||
|
||||
func writeRunResult(writer io.Writer, content []byte) error {
|
||||
for len(content) > 0 {
|
||||
written, err := writer.Write(content)
|
||||
if written < 0 || written > len(content) {
|
||||
return io.ErrShortWrite
|
||||
}
|
||||
content = content[written:]
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if written == 0 {
|
||||
return io.ErrShortWrite
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
199
internal/cli/run_result_command_test.go
Normal file
199
internal/cli/run_result_command_test.go
Normal file
@@ -0,0 +1,199 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
alwaysreject "gitea.maximumdirect.net/eric/notarius/internal/modules/generic/validate/always_reject"
|
||||
)
|
||||
|
||||
func TestMaintainedMinimalInvocationEmitsRunResult(t *testing.T) {
|
||||
outputRoot := filepath.Join(t.TempDir(), "output")
|
||||
var stdout, stderr strings.Builder
|
||||
code := RunWithOptions([]string{
|
||||
"run", "dnd-session",
|
||||
"--config", repositoryPath("examples", "dnd-minimal.config.yml"),
|
||||
"--input", repositoryPath("examples", "seriatim-minimal-transcript.json"),
|
||||
"--only", "spells", "--chunk_cache", "bypass", "--output-dir", outputRoot, "--json",
|
||||
}, &stdout, &stderr, productionRunOptions(t, &productionFakeLLMClient{}))
|
||||
if code != 0 || stderr.Len() != 0 {
|
||||
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||
}
|
||||
|
||||
receipt := decodeRunResultDocument(t, stdout.String())
|
||||
if got := receipt["schema_version"]; got != "notarius.run-result.v1" {
|
||||
t.Fatalf("schema_version = %q", got)
|
||||
}
|
||||
if got := receipt["run_id"]; got != productionRunID {
|
||||
t.Fatalf("run_id = %q", got)
|
||||
}
|
||||
if got := receipt["pipeline_id"]; got != "dnd-session" {
|
||||
t.Fatalf("pipeline_id = %q", got)
|
||||
}
|
||||
if got := receipt["index_file"]; got != "index.json" {
|
||||
t.Fatalf("index_file = %q", got)
|
||||
}
|
||||
if got := receipt["normalized_output_count"]; got != float64(1) {
|
||||
t.Fatalf("normalized_output_count = %v", got)
|
||||
}
|
||||
if got := receipt["rejected_output_count"]; got != float64(0) {
|
||||
t.Fatalf("rejected_output_count = %v", got)
|
||||
}
|
||||
if got := receipt["warning_count"]; got != float64(0) {
|
||||
t.Fatalf("warning_count = %v", got)
|
||||
}
|
||||
if got := receipt["validation_status"]; got != "approved" {
|
||||
t.Fatalf("validation_status = %q", got)
|
||||
}
|
||||
|
||||
outputDirectory, ok := receipt["output_directory"].(string)
|
||||
if !ok || !filepath.IsAbs(outputDirectory) || outputDirectory != filepath.Join(outputRoot, productionRunID) {
|
||||
t.Fatalf("output_directory = %q", receipt["output_directory"])
|
||||
}
|
||||
indexFile := receipt["index_file"].(string)
|
||||
assertFile(t, filepath.Join(outputDirectory, indexFile))
|
||||
}
|
||||
|
||||
func TestRunResultReportsWarningsAndDebugBundle(t *testing.T) {
|
||||
roots := newStateTestRoots(t)
|
||||
harness := newStateTestHarness()
|
||||
harness.chunkWarnings = []contracts.Warning{{Scope: "chunk", ReasonCode: "contract-warning", Message: "warning retained"}}
|
||||
var stdout, stderr bytes.Buffer
|
||||
code := RunWithOptions([]string{
|
||||
"run", "sample", "--config", roots.config, "--input", roots.input,
|
||||
"--chunk_cache", "bypass", "--debug", "--json",
|
||||
}, &stdout, &stderr, harness.options())
|
||||
if code != 0 || !strings.Contains(stderr.String(), "1 warning(s)") {
|
||||
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||
}
|
||||
|
||||
receipt := decodeRunResultDocument(t, stdout.String())
|
||||
if got := receipt["warning_count"]; got != float64(1) {
|
||||
t.Fatalf("warning_count = %v", got)
|
||||
}
|
||||
debugDirectory, ok := receipt["debug_directory"].(string)
|
||||
if !ok || !filepath.IsAbs(debugDirectory) || debugDirectory != onlyChildDir(t, roots.debug) {
|
||||
t.Fatalf("debug_directory = %q", receipt["debug_directory"])
|
||||
}
|
||||
if strings.Contains(stdout.String(), "complete:") || strings.Contains(stdout.String(), "debug=") {
|
||||
t.Fatalf("machine stdout contains human reporting: %q", stdout.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunResultReportsSuccessfulRejection(t *testing.T) {
|
||||
roots := newStateTestRoots(t)
|
||||
configBytes, err := os.ReadFile(roots.config)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
configBytes = []byte(replaceRequiredOnce(t, string(configBytes), " normalize: test/normalize\n", " normalize:\n module: test/normalize\n validators:\n - generic/always_reject\n"))
|
||||
if err := os.WriteFile(roots.config, configBytes, 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
harness := newStateTestHarness()
|
||||
opts := harness.options()
|
||||
if err := alwaysreject.RegisterTyped[stateTestArtifact](opts.Registries.Validators, stateTestArtifactKind); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
opts.Catalog = catalogFromRegistries(opts.Registries)
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
code := RunWithOptions([]string{
|
||||
"run", "sample", "--config", roots.config, "--input", roots.input,
|
||||
"--chunk_cache", "bypass", "--json",
|
||||
}, &stdout, &stderr, opts)
|
||||
if code != 0 || stderr.Len() != 0 {
|
||||
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||
}
|
||||
receipt := decodeRunResultDocument(t, stdout.String())
|
||||
if got := receipt["normalized_output_count"]; got != float64(0) {
|
||||
t.Fatalf("normalized_output_count = %v", got)
|
||||
}
|
||||
if got := receipt["rejected_output_count"]; got != float64(1) {
|
||||
t.Fatalf("rejected_output_count = %v", got)
|
||||
}
|
||||
if got := receipt["validation_status"]; got != "rejected" {
|
||||
t.Fatalf("validation_status = %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunResultIsAbsentForSyntaxAndRuntimeFailures(t *testing.T) {
|
||||
t.Run("syntax", func(t *testing.T) {
|
||||
var stdout, stderr bytes.Buffer
|
||||
code := RunWithOptions([]string{"run", "sample", "--json"}, &stdout, &stderr, newStateTestHarness().options())
|
||||
if code != 2 || stdout.Len() != 0 || stderr.Len() == 0 {
|
||||
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("runtime", func(t *testing.T) {
|
||||
roots := newStateTestRoots(t)
|
||||
harness := newStateTestHarness()
|
||||
harness.extractErr = errors.New("injected extraction failure")
|
||||
var stdout, stderr bytes.Buffer
|
||||
code := RunWithOptions([]string{
|
||||
"run", "sample", "--config", roots.config, "--input", roots.input,
|
||||
"--chunk_cache", "bypass", "--json",
|
||||
}, &stdout, &stderr, harness.options())
|
||||
if code != 1 || stdout.Len() != 0 || stderr.Len() == 0 {
|
||||
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestRunResultDeliveryFailureRetainsPublishedBundles(t *testing.T) {
|
||||
roots := newStateTestRoots(t)
|
||||
writerErr := errors.New("result writer sentinel")
|
||||
stdout := &resultDeliveryWriter{err: writerErr}
|
||||
var stderr bytes.Buffer
|
||||
code := RunWithOptions([]string{
|
||||
"run", "sample", "--config", roots.config, "--input", roots.input,
|
||||
"--chunk_cache", "bypass", "--debug", "--json",
|
||||
}, stdout, &stderr, newStateTestHarness().options())
|
||||
if code != 1 || !strings.Contains(stderr.String(), "write run result") || strings.Contains(stderr.String(), writerErr.Error()) {
|
||||
t.Fatalf("code=%d stderr=%q", code, stderr.String())
|
||||
}
|
||||
if stdout.accepted.Len() != 0 {
|
||||
t.Fatalf("accepted stdout = %q", stdout.accepted.String())
|
||||
}
|
||||
assertStateTestOutput(t, roots.output)
|
||||
debugBundle := onlyChildDir(t, roots.debug)
|
||||
report := readStateTestRunReport(t, debugBundle)
|
||||
if !report.Succeeded {
|
||||
t.Fatalf("debug report = %#v, want successful persisted run", report)
|
||||
}
|
||||
if strings.Contains(readAllFiles(t, debugBundle), writerErr.Error()) {
|
||||
t.Fatalf("debug bundle contains result writer error")
|
||||
}
|
||||
}
|
||||
|
||||
func decodeRunResultDocument(t *testing.T, stdout string) map[string]any {
|
||||
t.Helper()
|
||||
if strings.Count(stdout, "\n") != 1 {
|
||||
t.Fatalf("stdout = %q, want one JSON document", stdout)
|
||||
}
|
||||
var receipt map[string]any
|
||||
if err := json.Unmarshal([]byte(stdout), &receipt); err != nil {
|
||||
t.Fatalf("decode run result: %v; stdout=%q", err, stdout)
|
||||
}
|
||||
return receipt
|
||||
}
|
||||
|
||||
type resultDeliveryWriter struct {
|
||||
err error
|
||||
accepted bytes.Buffer
|
||||
}
|
||||
|
||||
func (w *resultDeliveryWriter) Write(content []byte) (int, error) {
|
||||
if w.err != nil {
|
||||
return 0, w.err
|
||||
}
|
||||
return w.accepted.Write(content)
|
||||
}
|
||||
200
internal/cli/run_result_test.go
Normal file
200
internal/cli/run_result_test.go
Normal file
@@ -0,0 +1,200 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"io"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
)
|
||||
|
||||
func TestRunResultEncodesRequiredFieldsAndCounts(t *testing.T) {
|
||||
result, err := newRunResult(testResolvedPipeline(pipeline.DefaultOutputModule), testRunOutput(), "relative-output", "relative-debug")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
encoded, err := encodeRunResult(result)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if encoded[len(encoded)-1] != '\n' || bytes.Count(encoded, []byte{'\n'}) != 1 {
|
||||
t.Fatalf("encoded result is not one newline-terminated object: %q", encoded)
|
||||
}
|
||||
|
||||
var decoded map[string]any
|
||||
if err := json.Unmarshal(encoded, &decoded); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := decoded["schema_version"]; got != runResultSchemaVersion {
|
||||
t.Fatalf("schema_version = %q", got)
|
||||
}
|
||||
if got := decoded["run_id"]; got != "run-123" {
|
||||
t.Fatalf("run_id = %q", got)
|
||||
}
|
||||
if got := decoded["pipeline_id"]; got != "sample" {
|
||||
t.Fatalf("pipeline_id = %q", got)
|
||||
}
|
||||
if got := decoded["validation_status"]; got != "rejected" {
|
||||
t.Fatalf("validation_status = %q", got)
|
||||
}
|
||||
if got := decoded["index_file"]; got != "index.json" {
|
||||
t.Fatalf("index_file = %q", got)
|
||||
}
|
||||
if got := decoded["normalized_output_count"]; got != float64(2) {
|
||||
t.Fatalf("normalized_output_count = %v", got)
|
||||
}
|
||||
if got := decoded["rejected_output_count"]; got != float64(1) {
|
||||
t.Fatalf("rejected_output_count = %v", got)
|
||||
}
|
||||
if got := decoded["warning_count"]; got != float64(1) {
|
||||
t.Fatalf("warning_count = %v", got)
|
||||
}
|
||||
if got := decoded["output_directory"]; got != filepath.Join(mustWorkingDirectory(t), "relative-output") {
|
||||
t.Fatalf("output_directory = %q", got)
|
||||
}
|
||||
if got := decoded["debug_directory"]; got != filepath.Join(mustWorkingDirectory(t), "relative-debug") {
|
||||
t.Fatalf("debug_directory = %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunResultRejectsInvalidRequiredValues(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
resolved pipeline.ResolvedPipeline
|
||||
output pipeline.RunOutput
|
||||
directory string
|
||||
}{
|
||||
{name: "blank run ID", resolved: testResolvedPipeline(pipeline.DefaultOutputModule), output: testRunOutputWithout(func(output *pipeline.RunOutput) { output.Manifest.RunID = " " }), directory: "output"},
|
||||
{name: "blank resolved pipeline ID", resolved: pipeline.ResolvedPipeline{Output: pipeline.ModuleBinding{Module: pipeline.DefaultOutputModule}}, output: testRunOutput(), directory: "output"},
|
||||
{name: "blank manifest pipeline ID", resolved: testResolvedPipeline(pipeline.DefaultOutputModule), output: testRunOutputWithout(func(output *pipeline.RunOutput) { output.Manifest.PipelineID = "" }), directory: "output"},
|
||||
{name: "mismatched pipeline IDs", resolved: testResolvedPipeline(pipeline.DefaultOutputModule), output: testRunOutputWithout(func(output *pipeline.RunOutput) { output.Manifest.PipelineID = "other" }), directory: "output"},
|
||||
{name: "blank validation status", resolved: testResolvedPipeline(pipeline.DefaultOutputModule), output: testRunOutputWithout(func(output *pipeline.RunOutput) { output.Manifest.ValidationStatus = " " }), directory: "output"},
|
||||
{name: "blank output directory", resolved: testResolvedPipeline(pipeline.DefaultOutputModule), output: testRunOutput(), directory: " "},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
if _, err := newRunResult(tt.resolved, tt.output, tt.directory, ""); err == nil {
|
||||
t.Fatal("newRunResult() succeeded")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunResultOmitsIndexFileForOtherOutputModules(t *testing.T) {
|
||||
result, err := newRunResult(testResolvedPipeline("test/output"), testRunOutputWithout(func(output *pipeline.RunOutput) {
|
||||
output.OutputFiles = nil
|
||||
}), "output", "")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if result.IndexFile != "" {
|
||||
t.Fatalf("index_file = %q", result.IndexFile)
|
||||
}
|
||||
encoded, err := encodeRunResult(result)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var decoded map[string]any
|
||||
if err := json.Unmarshal(encoded, &decoded); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, ok := decoded["index_file"]; ok {
|
||||
t.Fatalf("encoded non-JSON result contains index_file: %s", encoded)
|
||||
}
|
||||
if _, ok := decoded["debug_directory"]; ok {
|
||||
t.Fatalf("encoded result without debug capture contains debug_directory: %s", encoded)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunResultRequiresOneProductionIndexFile(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
files []contracts.OutputFile
|
||||
}{
|
||||
{name: "missing", files: nil},
|
||||
{name: "duplicate", files: []contracts.OutputFile{{Name: "index.json"}, {Name: "index.json"}}},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
output := testRunOutput()
|
||||
output.OutputFiles = tt.files
|
||||
if _, err := newRunResult(testResolvedPipeline(pipeline.DefaultOutputModule), output, "output", ""); err == nil {
|
||||
t.Fatal("newRunResult() succeeded")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriteRunResultCompletesAndReportsWriterFailure(t *testing.T) {
|
||||
content := []byte("result\n")
|
||||
var target bytes.Buffer
|
||||
if err := writeRunResult(partialResultWriter{writer: &target, limit: 2}, content); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := target.String(); got != string(content) {
|
||||
t.Fatalf("written result = %q", got)
|
||||
}
|
||||
|
||||
writerErr := errors.New("result writer failed")
|
||||
if err := writeRunResult(failingResultWriter{err: writerErr}, content); !errors.Is(err, writerErr) {
|
||||
t.Fatalf("writeRunResult() error = %v", err)
|
||||
}
|
||||
if err := writeRunResult(zeroResultWriter{}, content); !errors.Is(err, io.ErrShortWrite) {
|
||||
t.Fatalf("zero-progress error = %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func testResolvedPipeline(outputModule string) pipeline.ResolvedPipeline {
|
||||
return pipeline.ResolvedPipeline{ID: "sample", Output: pipeline.ModuleBinding{Module: outputModule}}
|
||||
}
|
||||
|
||||
func testRunOutput() pipeline.RunOutput {
|
||||
return pipeline.RunOutput{
|
||||
Manifest: artifacts.RunManifest{RunID: "run-123", PipelineID: "sample", ValidationStatus: "rejected"},
|
||||
NormalizeOutputs: []contracts.SerializedOutput{{}, {}},
|
||||
Rejected: []contracts.RejectedOutput{{}},
|
||||
Warnings: []contracts.Warning{{}},
|
||||
OutputFiles: []contracts.OutputFile{{Name: "index.json"}},
|
||||
}
|
||||
}
|
||||
|
||||
func testRunOutputWithout(change func(*pipeline.RunOutput)) pipeline.RunOutput {
|
||||
output := testRunOutput()
|
||||
change(&output)
|
||||
return output
|
||||
}
|
||||
|
||||
func mustWorkingDirectory(t *testing.T) string {
|
||||
t.Helper()
|
||||
workingDirectory, err := filepath.Abs(".")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return workingDirectory
|
||||
}
|
||||
|
||||
type partialResultWriter struct {
|
||||
writer io.Writer
|
||||
limit int
|
||||
}
|
||||
|
||||
func (w partialResultWriter) Write(content []byte) (int, error) {
|
||||
if len(content) > w.limit {
|
||||
content = content[:w.limit]
|
||||
}
|
||||
return w.writer.Write(content)
|
||||
}
|
||||
|
||||
type failingResultWriter struct{ err error }
|
||||
|
||||
func (w failingResultWriter) Write([]byte) (int, error) { return 0, w.err }
|
||||
|
||||
type zeroResultWriter struct{}
|
||||
|
||||
func (zeroResultWriter) Write([]byte) (int, error) { return 0, nil }
|
||||
@@ -1,68 +0,0 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"testing/fstest"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||
"gitea.maximumdirect.net/eric/scriptorium"
|
||||
)
|
||||
|
||||
const profileCheckPromptID = "notarius.profile.check"
|
||||
|
||||
var profileCheckPromptFS = fstest.MapFS{
|
||||
"prompts/profile-check.yaml": &fstest.MapFile{Data: []byte(`id: notarius.profile.check
|
||||
version: "1.0.0"
|
||||
default_profile: mistral-small-3
|
||||
inputs:
|
||||
- name: transcript
|
||||
required: true
|
||||
messages:
|
||||
- role: user
|
||||
content: "{{input \"transcript\"}}"
|
||||
output:
|
||||
format: text
|
||||
validation_mode: none
|
||||
repair_attempts: 0
|
||||
`)},
|
||||
}
|
||||
|
||||
func validateExplicitScriptoriumProfiles(ctx context.Context, cfg config.Config, profileIDs []string) error {
|
||||
if len(profileIDs) == 0 {
|
||||
return nil
|
||||
}
|
||||
engine, err := newProfileValidationEngine(cfg)
|
||||
if err != nil {
|
||||
return fmt.Errorf("load Scriptorium profiles: %w", err)
|
||||
}
|
||||
for _, profileID := range profileIDs {
|
||||
if _, err := engine.Prepare(ctx, scriptorium.RunRequest{
|
||||
PromptID: profileCheckPromptID,
|
||||
ProfileID: profileID,
|
||||
Inputs: map[string]scriptorium.ArtifactRef{
|
||||
"transcript": scriptorium.Inline("profile check"),
|
||||
},
|
||||
}); err != nil {
|
||||
if errors.Is(err, scriptorium.ErrProfileNotFound) {
|
||||
return fmt.Errorf("Scriptorium profile %q is not configured", profileID)
|
||||
}
|
||||
return fmt.Errorf("validate Scriptorium profile %q: %w", profileID, err)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func newProfileValidationEngine(cfg config.Config) (*scriptorium.Engine, error) {
|
||||
opts := []scriptorium.Option{
|
||||
scriptorium.WithPromptFS(profileCheckPromptFS, "prompts"),
|
||||
}
|
||||
if cfg.Scriptorium.ProfileFile != "" {
|
||||
opts = append(opts, scriptorium.WithProfileFile(cfg.Scriptorium.ProfileFile))
|
||||
}
|
||||
return scriptorium.NewEngine(scriptorium.Config{
|
||||
PromptDir: "unused",
|
||||
ProfileDir: cfg.Scriptorium.ProfileDir,
|
||||
}, opts...)
|
||||
}
|
||||
@@ -24,7 +24,7 @@ import (
|
||||
|
||||
func TestSpellCatalogBytesAffectCheckpointIdentityButNotSemanticDigest(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
configPath := repositoryPath("examples", "dnd-spells-production.config.yml")
|
||||
configPath := writeProductionSpellCatalogContractConfig(t)
|
||||
effective, err := loadMaintainedExample(t, configPath).Resolve(resolveInputForMaintainedExample(components, "dnd-session"))
|
||||
if err != nil {
|
||||
t.Fatalf("resolve production configuration: %v", err)
|
||||
@@ -108,8 +108,8 @@ func TestSpellCatalogBytesAffectCheckpointIdentityButNotSemanticDigest(t *testin
|
||||
}
|
||||
|
||||
func TestConfiguredSpellCatalogBindingChangesResolvedPipelineIdentity(t *testing.T) {
|
||||
base := string(readRepositoryFile(t, "examples", "dnd-spells-production.config.yml"))
|
||||
changed := strings.Replace(base, "./dnd-spells-catalog.json", "./alternate-spell-catalog.json", 1)
|
||||
base := productionSpellCatalogContractConfig(t)
|
||||
changed := strings.Replace(base, repositoryPath("examples", "dnd-spell-catalog.json"), filepath.Join(t.TempDir(), "alternate-spell-catalog.json"), 1)
|
||||
if changed == base {
|
||||
t.Fatal("production configuration did not contain the maintained catalog binding")
|
||||
}
|
||||
@@ -138,7 +138,7 @@ func TestConfiguredSpellCatalogBindingChangesResolvedPipelineIdentity(t *testing
|
||||
|
||||
func TestSemanticSpellCatalogFingerprintChangesCheckpointIdentityWithoutReferenceChange(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
configPath := repositoryPath("examples", "dnd-spells-production.config.yml")
|
||||
configPath := writeProductionSpellCatalogContractConfig(t)
|
||||
effective, err := loadMaintainedExample(t, configPath).Resolve(resolveInputForMaintainedExample(components, "dnd-session"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
@@ -154,9 +154,9 @@ func TestSemanticSpellCatalogFingerprintChangesCheckpointIdentityWithoutReferenc
|
||||
fingerprints := prepared.CheckpointFingerprints()
|
||||
wantNames := map[string]struct{}{
|
||||
"extract:spells:" + spells.Key + ":effective_catalog": {},
|
||||
"extract:spells:" + spells.Key + ":validator:3:extract/dnd/spells/catalog:effective_catalog": {},
|
||||
"extract:spells:" + spells.Key + ":validator:2:extract/dnd/spells/catalog:effective_catalog": {},
|
||||
"normalize:spells:" + spellnormalize.Key + ":effective_catalog": {},
|
||||
"normalize:spells:" + spellnormalize.Key + ":validator:3:extract/dnd/spells/catalog:effective_catalog": {},
|
||||
"normalize:spells:" + spellnormalize.Key + ":validator:2:extract/dnd/spells/catalog:effective_catalog": {},
|
||||
}
|
||||
seen := make(map[string]string, len(fingerprints))
|
||||
for _, fingerprint := range fingerprints {
|
||||
@@ -200,7 +200,7 @@ func TestSemanticSpellCatalogFingerprintChangesCheckpointIdentityWithoutReferenc
|
||||
|
||||
func TestChangedSemanticSpellCatalogFingerprintCannotResumeRecordedCheckpoint(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
configPath := repositoryPath("examples", "dnd-spells-production.config.yml")
|
||||
configPath := writeProductionSpellCatalogContractConfig(t)
|
||||
effective, err := loadMaintainedExample(t, configPath).Resolve(resolveInputForMaintainedExample(components, "dnd-session"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
@@ -214,8 +214,9 @@ func TestChangedSemanticSpellCatalogFingerprintCannotResumeRecordedCheckpoint(t
|
||||
t.Fatal(err)
|
||||
}
|
||||
fingerprints := prepared.CheckpointFingerprints()
|
||||
llmFingerprints := []checkpoint.Fingerprint{{Name: "promptkit_profile_source", Value: "sha256:profile-source-one"}}
|
||||
settings := config.CheckpointCacheConfig{Enabled: true, Directory: t.TempDir()}
|
||||
recorder, _, err := checkpointHandlersForRun(settings, Options{}, materialized, fingerprints, []byte("same input"), nil, nil, "", "", false)
|
||||
recorder, _, err := checkpointHandlersForRun(settings, Options{}, materialized, fingerprints, llmFingerprints, []byte("same input"), nil, nil, "", "", LLMRuntimeOverrides{}, false)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
@@ -241,7 +242,7 @@ func TestChangedSemanticSpellCatalogFingerprintCannotResumeRecordedCheckpoint(t
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
_, sameLoader, err := checkpointHandlersForRun(settings, Options{}, materialized, fingerprints, []byte("same input"), nil, nil, "", "", true)
|
||||
_, sameLoader, err := checkpointHandlersForRun(settings, Options{}, materialized, fingerprints, llmFingerprints, []byte("same input"), nil, nil, "", "", LLMRuntimeOverrides{}, true)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
@@ -253,7 +254,7 @@ func TestChangedSemanticSpellCatalogFingerprintCannotResumeRecordedCheckpoint(t
|
||||
}
|
||||
changed := replaceCheckpointFingerprintValue(t, fingerprints, normalizeSpellCatalogFingerprintName(), "sha256:changed-effective-catalog")
|
||||
assertOnlyCheckpointFingerprintChanged(t, fingerprints, changed, normalizeSpellCatalogFingerprintName())
|
||||
_, changedLoader, err := checkpointHandlersForRun(settings, Options{}, materialized, changed, []byte("same input"), nil, nil, "", "", true)
|
||||
_, changedLoader, err := checkpointHandlersForRun(settings, Options{}, materialized, changed, llmFingerprints, []byte("same input"), nil, nil, "", "", LLMRuntimeOverrides{}, true)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
@@ -263,12 +264,37 @@ func TestChangedSemanticSpellCatalogFingerprintCannotResumeRecordedCheckpoint(t
|
||||
if _, decision := changedLoader.Normalize("spells", spellnormalize.Key, normalizeDependencies); decision.Reused {
|
||||
t.Fatalf("changed normalize fingerprint decision = %#v, want normalize checkpoint cold miss", decision)
|
||||
}
|
||||
changedMapping := replaceCheckpointFingerprintValue(t, fingerprints, extractSpellMappingFingerprintName(), "dnd.spells.extract_mapping.v3")
|
||||
assertOnlyCheckpointFingerprintChanged(t, fingerprints, changedMapping, extractSpellMappingFingerprintName())
|
||||
_, mappingLoader, err := checkpointHandlersForRun(settings, Options{}, materialized, changedMapping, llmFingerprints, []byte("same input"), nil, nil, "", "", LLMRuntimeOverrides{}, true)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, decision := mappingLoader.Source(materialized.Input.Module); decision.Reused {
|
||||
t.Fatalf("changed mapping policy decision = %#v, want cold miss", decision)
|
||||
}
|
||||
|
||||
changedLLMFingerprints := []checkpoint.Fingerprint{{Name: "promptkit_profile_source", Value: "sha256:profile-source-two"}}
|
||||
_, profileLoader, err := checkpointHandlersForRun(settings, Options{}, materialized, fingerprints, changedLLMFingerprints, []byte("same input"), nil, nil, "", "", LLMRuntimeOverrides{}, true)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, decision := profileLoader.Source(materialized.Input.Module); decision.Reused {
|
||||
t.Fatalf("changed PromptKit profile source decision = %#v, want cold miss", decision)
|
||||
}
|
||||
if _, decision := profileLoader.Normalize("spells", spellnormalize.Key, normalizeDependencies); decision.Reused {
|
||||
t.Fatalf("changed PromptKit profile normalize decision = %#v, want cold miss", decision)
|
||||
}
|
||||
}
|
||||
|
||||
func normalizeSpellCatalogFingerprintName() string {
|
||||
return "normalize:spells:" + spellnormalize.Key + ":effective_catalog"
|
||||
}
|
||||
|
||||
func extractSpellMappingFingerprintName() string {
|
||||
return "extract:spells:" + spells.Key + ":mapping_policy"
|
||||
}
|
||||
|
||||
func replaceCheckpointFingerprintValue(t *testing.T, fingerprints []pipeline.CheckpointFingerprint, name, value string) []pipeline.CheckpointFingerprint {
|
||||
t.Helper()
|
||||
changed := append([]pipeline.CheckpointFingerprint(nil), fingerprints...)
|
||||
@@ -315,7 +341,7 @@ func TestMaintainedProductionOverlayRunAlignsGroundingValidationAndProvenance(t
|
||||
var stdout, stderr strings.Builder
|
||||
code := RunWithOptions([]string{
|
||||
"run", "dnd-session",
|
||||
"--config", repositoryPath("examples", "dnd-spells-production.config.yml"),
|
||||
"--config", writeProductionSpellCatalogContractConfig(t),
|
||||
"--input", repositoryPath("examples", "seriatim-minimal-transcript.json"),
|
||||
"--only", "spells", "--chunk_cache", "bypass", "--output-dir", outputRoot,
|
||||
}, &stdout, &stderr, options)
|
||||
@@ -360,12 +386,12 @@ func TestMaintainedProductionOverlayRunAlignsGroundingValidationAndProvenance(t
|
||||
if len(catalogProvenances) != 2 {
|
||||
t.Fatalf("manifest references = %#v, want independently materialized extract and normalize catalog provenance", manifest.References)
|
||||
}
|
||||
overlayBytes := readRepositoryFile(t, "examples", "dnd-spells-catalog.json")
|
||||
overlayBytes := readRepositoryFile(t, "examples", "dnd-spell-catalog.json")
|
||||
for _, catalogProvenance := range catalogProvenances {
|
||||
if catalogProvenance.Stage != "extract" && catalogProvenance.Stage != "normalize" {
|
||||
t.Fatalf("catalog provenance = %#v, want extract or normalize scope", catalogProvenance)
|
||||
}
|
||||
if catalogProvenance.LaneID != "spells" || catalogProvenance.OriginType != "file" || catalogProvenance.MediaType != "application/json" || catalogProvenance.SizeBytes != int64(len(overlayBytes)) || catalogProvenance.Digest != digestBytes(overlayBytes) || !strings.Contains(catalogProvenance.OriginURI, "dnd-spells-catalog.json") {
|
||||
if catalogProvenance.LaneID != "spells" || catalogProvenance.OriginType != "file" || catalogProvenance.MediaType != "application/json" || catalogProvenance.SizeBytes != int64(len(overlayBytes)) || catalogProvenance.Digest != digestBytes(overlayBytes) || !strings.Contains(catalogProvenance.OriginURI, "dnd-spell-catalog.json") {
|
||||
t.Fatalf("catalog provenance = %#v, want raw overlay provenance in both scopes", catalogProvenance)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"path/filepath"
|
||||
"sync"
|
||||
"testing"
|
||||
|
||||
@@ -50,7 +51,7 @@ func TestProductionSpellCatalogValidationRetries(t *testing.T) {
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
components := productionTestComponents(t)
|
||||
configPath := repositoryPath("examples", "dnd-spells-production.config.yml")
|
||||
configPath := writeProductionSpellCatalogContractConfig(t)
|
||||
cfg := loadMaintainedExample(t, configPath)
|
||||
effective, err := cfg.Resolve(config.ResolveInput{PipelineID: "dnd-session", Catalog: catalogFromRegistries(components.registries)})
|
||||
if err != nil {
|
||||
@@ -58,7 +59,7 @@ func TestProductionSpellCatalogValidationRetries(t *testing.T) {
|
||||
}
|
||||
materialized, _, err := pipeline.MaterializeReferences(effective.ResolvedPipeline, catalogFromRegistries(components.registries), pipeline.ReferenceMaterializationOptions{
|
||||
ConfigPath: configPath,
|
||||
WorkingDir: repositoryPath("examples"),
|
||||
WorkingDir: filepath.Dir(configPath),
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("materialize production references: %v", err)
|
||||
@@ -150,8 +151,6 @@ func productionSpellResponse(name string) string {
|
||||
content, err := json.Marshal(dnd.SpellList{SpellCasts: []dnd.SpellCast{{
|
||||
Caster: "Aria",
|
||||
Spell: name,
|
||||
Effect: "The spell takes effect.",
|
||||
NarrativeDescription: "Aria casts the spell.",
|
||||
SourceRefs: []source.SourceRef{{SourceID: "session-alpha", StartUnitID: 1, EndUnitID: 1}},
|
||||
}}})
|
||||
if err != nil {
|
||||
|
||||
@@ -24,7 +24,7 @@ import (
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
)
|
||||
|
||||
const stateTestDigest = "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"
|
||||
const stateTestDigest = "sha256:e511d8906649b78eb639b11215fa57a9652a1a64f4aefa3ed68320dbda46f439"
|
||||
|
||||
func TestRunStateSurfaceMatrix(t *testing.T) {
|
||||
for _, debug := range []bool{false, true} {
|
||||
@@ -635,7 +635,7 @@ func newStateTestRoots(t *testing.T) stateTestRoots {
|
||||
t.Fatal(err)
|
||||
}
|
||||
roots.config = filepath.Join(base, "config.yml")
|
||||
config := fmt.Sprintf("version: 3\noutput:\n directory: %q\ncache:\n chunk_plans:\n directory: %q\n mode: auto\n checkpoints:\n enabled: true\n directory: %q\ndebug:\n directory: %q\npipelines:\n sample:\n input: test/input\n chunk: test/chunk\n artifacts:\n items:\n extract: test/extract\n merge: test/merge\n normalize: test/normalize\n output: test/output\n", roots.output, roots.plans, roots.checkpoints, roots.debug)
|
||||
config := fmt.Sprintf("version: 4\noutput:\n directory: %q\ncache:\n chunk_plans:\n directory: %q\n mode: auto\n checkpoints:\n enabled: true\n directory: %q\ndebug:\n directory: %q\npipelines:\n sample:\n input: test/input\n chunk: test/chunk\n artifacts:\n items:\n extract: test/extract\n merge: test/merge\n normalize: test/normalize\n output: test/output\n", roots.output, roots.plans, roots.checkpoints, roots.debug)
|
||||
if err := os.WriteFile(roots.config, []byte(config), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
@@ -823,22 +823,22 @@ func (h *stateTestHarness) options() Options {
|
||||
if err := pipeline.RegisterArtifactCodec(registries.ArtifactCodecs, stateTestCodec{}); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
if err := registries.Inputs.RegisterBuilderWithSpec(pipeline.ModuleSpec{Key: "test/input", Stage: pipeline.StageInput, Provides: []string{"source"}}, func(map[string]any) error { return nil }, func(pipeline.BuildRequest) (contracts.InputAdapter, error) { return stateTestInput{}, nil }); err != nil {
|
||||
if err := registries.Inputs.RegisterBuilderWithSpec(pipeline.ModuleSpec{Key: "test/input", Stage: pipeline.StageInput, ExecutionClass: contracts.ExecutionClassDeterministic, Provides: []string{"source"}}, func(map[string]any) error { return nil }, func(pipeline.BuildRequest) (contracts.InputAdapter, error) { return stateTestInput{}, nil }); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
if err := registries.Chunkers.RegisterBuilderWithSpec(pipeline.ModuleSpec{Key: "test/chunk", Stage: pipeline.StageChunk, Requires: []string{"source"}, Provides: []string{"chunks"}, ReferenceSlots: []contracts.ReferenceSlot{{Name: "cache-reference"}}}, func(map[string]any) error { return nil }, func(pipeline.BuildRequest) (contracts.Chunker, error) { return stateTestChunker{h}, nil }); err != nil {
|
||||
if err := registries.Chunkers.RegisterBuilderWithSpec(pipeline.ModuleSpec{Key: "test/chunk", Stage: pipeline.StageChunk, ExecutionClass: contracts.ExecutionClassLLMBacked, Requires: []string{"source"}, Provides: []string{"chunks"}, ReferenceSlots: []contracts.ReferenceSlot{{Name: "cache-reference"}}}, func(map[string]any) error { return nil }, func(pipeline.BuildRequest) (contracts.Chunker, error) { return stateTestChunker{h}, nil }); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
if err := pipeline.RegisterExtractor(registries.Extractors, pipeline.ModuleSpec{Key: "test/extract", Stage: pipeline.StageExtract, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: stateTestArtifactKind}, func() (contracts.Extractor[stateTestArtifact], error) { return stateTestExtractor{h}, nil }); err != nil {
|
||||
if err := pipeline.RegisterExtractor(registries.Extractors, pipeline.ModuleSpec{Key: "test/extract", Stage: pipeline.StageExtract, ExecutionClass: contracts.ExecutionClassLLMBacked, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: stateTestArtifactKind}, func() (contracts.Extractor[stateTestArtifact], error) { return stateTestExtractor{h}, nil }); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
if err := pipeline.RegisterMerger(registries.Mergers, pipeline.ModuleSpec{Key: "test/merge", Stage: pipeline.StageMerge, Requires: []string{"artifact"}, Provides: []string{"merged"}, ArtifactKind: stateTestArtifactKind}, func() (contracts.Merger[stateTestArtifact], error) { return stateTestMerger{harness: h}, nil }); err != nil {
|
||||
if err := pipeline.RegisterMerger(registries.Mergers, pipeline.ModuleSpec{Key: "test/merge", Stage: pipeline.StageMerge, ExecutionClass: contracts.ExecutionClassLLMBacked, Requires: []string{"artifact"}, Provides: []string{"merged"}, ArtifactKind: stateTestArtifactKind}, func() (contracts.Merger[stateTestArtifact], error) { return stateTestMerger{harness: h}, nil }); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
if err := pipeline.RegisterNormalizer(registries.Normalizers, pipeline.ModuleSpec{Key: "test/normalize", Stage: pipeline.StageNormalize, Requires: []string{"merged"}, Provides: []string{"normalized"}, ArtifactKind: stateTestArtifactKind}, func() (contracts.Normalizer[stateTestArtifact], error) { return stateTestNormalizer{harness: h}, nil }); err != nil {
|
||||
if err := pipeline.RegisterNormalizer(registries.Normalizers, pipeline.ModuleSpec{Key: "test/normalize", Stage: pipeline.StageNormalize, ExecutionClass: contracts.ExecutionClassLLMBacked, Requires: []string{"merged"}, Provides: []string{"normalized"}, ArtifactKind: stateTestArtifactKind}, func() (contracts.Normalizer[stateTestArtifact], error) { return stateTestNormalizer{harness: h}, nil }); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
if err := registries.Outputs.RegisterWithSpec(pipeline.ModuleSpec{Key: "test/output", Stage: pipeline.StageOutput, Requires: []string{"normalized"}, Provides: []string{"output"}}, func() (contracts.OutputEncoder, error) {
|
||||
if err := registries.Outputs.RegisterWithSpec(pipeline.ModuleSpec{Key: "test/output", Stage: pipeline.StageOutput, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"normalized"}, Provides: []string{"output"}}, func() (contracts.OutputEncoder, error) {
|
||||
return stateTestOutput{harness: h, includeWarnings: h.includeWarnings}, nil
|
||||
}); err != nil {
|
||||
panic(err)
|
||||
@@ -848,7 +848,7 @@ func (h *stateTestHarness) options() Options {
|
||||
defer h.mu.Unlock()
|
||||
h.runIDCalls++
|
||||
return fmt.Sprintf("run-%d-%032x", startedAt.UnixNano(), h.runIDCalls), nil
|
||||
}, UserCacheDir: func() (string, error) { return "", errors.New("unexpected user cache lookup") }, LLMClientFactory: func(context.Context, config.Config, string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
}, UserCacheDir: func() (string, error) { return "", errors.New("unexpected user cache lookup") }, LLMClientFactory: func(context.Context, config.Config, string, LLMRuntimeOverrides) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||
return nil, nil, nil
|
||||
}}
|
||||
}
|
||||
@@ -857,7 +857,13 @@ type stateTestInput struct{}
|
||||
|
||||
func (stateTestInput) Key() string { return "test/input" }
|
||||
func (stateTestInput) Parse(_ context.Context, req contracts.ParseRequest) (*source.SourceDocument, error) {
|
||||
return &source.SourceDocument{ID: "source", Kind: "text", Format: "text/plain", Digest: stateTestDigest, Units: []source.SourceUnit{{ID: 1, Kind: "text", Text: string(req.Raw), Ref: source.SourceRef{SourceID: "source", StartUnitID: 1, EndUnitID: 1}}}}, nil
|
||||
doc := &source.SourceDocument{ID: "source", Kind: "text", Format: "text/plain", Units: []source.SourceUnit{{ID: 1, Kind: "text", Text: string(req.Raw), Ref: source.SourceRef{SourceID: "source", StartUnitID: 1, EndUnitID: 1}}}}
|
||||
digest, err := source.DigestDocument(doc)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
doc.Digest = digest
|
||||
return doc, nil
|
||||
}
|
||||
|
||||
type stateTestChunker struct{ harness *stateTestHarness }
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
package artifacts
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
@@ -29,6 +30,29 @@ type LLMProfileManifest struct {
|
||||
ID string `json:"id"`
|
||||
Provider string `json:"provider,omitempty"`
|
||||
Model string `json:"model,omitempty"`
|
||||
BackendID string `json:"backend_id,omitempty"`
|
||||
ReasoningEffort string `json:"reasoning_effort,omitempty"`
|
||||
}
|
||||
|
||||
// Normalized returns the canonical representation used for manifest identity
|
||||
// and publication.
|
||||
func (profile LLMProfileManifest) Normalized() LLMProfileManifest {
|
||||
profile.ID = strings.TrimSpace(profile.ID)
|
||||
profile.Provider = strings.TrimSpace(profile.Provider)
|
||||
profile.Model = strings.TrimSpace(profile.Model)
|
||||
profile.BackendID = strings.TrimSpace(profile.BackendID)
|
||||
profile.ReasoningEffort = strings.TrimSpace(profile.ReasoningEffort)
|
||||
return profile
|
||||
}
|
||||
|
||||
// IdentityKey returns an opaque, deterministic key for the effective profile.
|
||||
func (profile LLMProfileManifest) IdentityKey() string {
|
||||
profile = profile.Normalized()
|
||||
return profile.ID + "\x00" +
|
||||
profile.Provider + "\x00" +
|
||||
profile.Model + "\x00" +
|
||||
profile.BackendID + "\x00" +
|
||||
profile.ReasoningEffort
|
||||
}
|
||||
|
||||
type ReferenceProvenance struct {
|
||||
|
||||
@@ -20,7 +20,7 @@ func TestRunManifestOmitsEmptyOptionalFields(t *testing.T) {
|
||||
func TestRunManifestChunkPlanIsAdditiveAndOmitsPlanContent(t *testing.T) {
|
||||
manifest := RunManifest{ChunkPlan: &ChunkPlanManifest{
|
||||
Mode: "auto", Action: "reused", SourceDigest: "sha256:source", PlanDigest: "sha256:plan",
|
||||
PlanSchemaVersion: "notarius.chunk-plan.v1", RequestedModule: "chunk/current",
|
||||
PlanSchemaVersion: "notarius.chunk-plan.v2", RequestedModule: "chunk/current",
|
||||
ProducerInputModule: "input/original", ProducerModule: "chunk/original",
|
||||
}}
|
||||
encoded, err := json.Marshal(manifest)
|
||||
@@ -53,7 +53,13 @@ func TestRunManifestIncludesPipelineAndArtifactLaneFields(t *testing.T) {
|
||||
PipelineID: "pipeline-1",
|
||||
PipelineDigest: "sha256:abc123",
|
||||
LLMProfiles: []LLMProfileManifest{
|
||||
{ID: "default", Provider: "scriptorium", Model: "model-a"},
|
||||
{
|
||||
ID: "default",
|
||||
Provider: "promptkit",
|
||||
Model: "model-a",
|
||||
BackendID: "openrouter",
|
||||
ReasoningEffort: "high",
|
||||
},
|
||||
},
|
||||
ArtifactLanes: []ArtifactLaneManifest{
|
||||
{
|
||||
@@ -101,7 +107,13 @@ func TestRunManifestIncludesPipelineAndArtifactLaneFields(t *testing.T) {
|
||||
if !ok {
|
||||
t.Fatalf("llm_profiles[0] = %#v, want object", profiles[0])
|
||||
}
|
||||
assertHasKeys(t, profile, "id", "provider", "model")
|
||||
assertHasKeys(t, profile, "id", "provider", "model", "backend_id", "reasoning_effort")
|
||||
if profile["provider"] != "promptkit" {
|
||||
t.Fatalf("llm_profiles[0].provider = %#v, want promptkit", profile["provider"])
|
||||
}
|
||||
if profile["backend_id"] != "openrouter" || profile["reasoning_effort"] != "high" {
|
||||
t.Fatalf("llm_profiles[0] = %#v, want backend and reasoning provenance", profile)
|
||||
}
|
||||
|
||||
lanes, ok := got["artifact_lanes"].([]any)
|
||||
if !ok {
|
||||
|
||||
@@ -4,10 +4,10 @@ import (
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
)
|
||||
|
||||
const SupportedFileConfigVersion = 3
|
||||
const SupportedFileConfigVersion = 4
|
||||
|
||||
type Config struct {
|
||||
Scriptorium ScriptoriumConfig `json:"scriptorium,omitempty"`
|
||||
PromptKit PromptKitConfig `json:"promptkit,omitempty"`
|
||||
Pipelines map[string]pipeline.PipelineProfile `json:"pipelines"`
|
||||
Concurrency ConcurrencyConfig `json:"concurrency"`
|
||||
Output OutputConfig `json:"output"`
|
||||
@@ -15,9 +15,15 @@ type Config struct {
|
||||
Debug DebugConfig `json:"debug"`
|
||||
}
|
||||
|
||||
type ScriptoriumConfig struct {
|
||||
type PromptKitConfig struct {
|
||||
ProfileDir string `json:"profile_dir,omitempty"`
|
||||
ProfileFile string `json:"profile_file,omitempty"`
|
||||
LocalBackend *PromptKitLocalBackendConfig `json:"local_backend,omitempty"`
|
||||
}
|
||||
|
||||
type PromptKitLocalBackendConfig struct {
|
||||
Endpoint string `json:"endpoint"`
|
||||
ConcurrencyLimit int `json:"concurrency_limit"`
|
||||
}
|
||||
|
||||
type ConcurrencyConfig struct {
|
||||
@@ -66,6 +72,10 @@ func Default() Config {
|
||||
|
||||
func cloneConfig(in Config) Config {
|
||||
out := in
|
||||
if in.PromptKit.LocalBackend != nil {
|
||||
localBackend := *in.PromptKit.LocalBackend
|
||||
out.PromptKit.LocalBackend = &localBackend
|
||||
}
|
||||
out.Concurrency.StageWorkers = cloneIntMap(in.Concurrency.StageWorkers)
|
||||
out.Pipelines = make(map[string]pipeline.PipelineProfile, len(in.Pipelines))
|
||||
for key, profile := range in.Pipelines {
|
||||
|
||||
@@ -42,12 +42,10 @@ func (c Config) Resolve(input ResolveInput) (EffectiveConfig, error) {
|
||||
}
|
||||
profile = clonePipelineProfile(profile)
|
||||
profile.ID = pipelineID
|
||||
if override := strings.TrimSpace(input.LLMProfileOverride); override != "" {
|
||||
applyLLMProfileOverride(&profile, override)
|
||||
}
|
||||
|
||||
resolved, err := pipeline.ResolvePipeline(profile, pipeline.ResolveOptions{
|
||||
Only: input.Only,
|
||||
LLMProfileOverride: input.LLMProfileOverride,
|
||||
ReferenceOverrides: append([]pipeline.ReferenceBinding(nil), input.ReferenceOverrides...),
|
||||
ReferenceUnbinds: append([]pipeline.ReferenceUnbind(nil), input.ReferenceUnbinds...),
|
||||
}, input.Catalog)
|
||||
@@ -65,22 +63,6 @@ func (c Config) Resolve(input ResolveInput) (EffectiveConfig, error) {
|
||||
}, nil
|
||||
}
|
||||
|
||||
func applyLLMProfileOverride(profile *pipeline.PipelineProfile, profileID string) {
|
||||
profile.Chunk.LLMProfile = profileID
|
||||
apply := func(artifacts map[string]pipeline.ArtifactLaneProfile) {
|
||||
for laneID, lane := range artifacts {
|
||||
lane.Extract.LLMProfile = profileID
|
||||
lane.Merge.LLMProfile = profileID
|
||||
lane.Normalize.LLMProfile = profileID
|
||||
artifacts[laneID] = lane
|
||||
}
|
||||
}
|
||||
apply(profile.Artifacts)
|
||||
for index := range profile.Steps {
|
||||
apply(profile.Steps[index].Artifacts)
|
||||
}
|
||||
}
|
||||
|
||||
func lookupPipelineProfile(profiles map[string]pipeline.PipelineProfile, pipelineID string) (pipeline.PipelineProfile, bool) {
|
||||
pipelineID = strings.TrimSpace(pipelineID)
|
||||
for rawID, profile := range profiles {
|
||||
|
||||
@@ -84,6 +84,56 @@ func TestEffectiveConfigMaterializesDefaultBindingsThroughCatalog(t *testing.T)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEffectiveConfigPreservesPromptKitProfileSource(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
profileSource PromptKitConfig
|
||||
}{
|
||||
{name: "profile directory", profileSource: PromptKitConfig{ProfileDir: "./profiles"}},
|
||||
{name: "profile file", profileSource: PromptKitConfig{ProfileFile: "./profiles.yml"}},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
cfg := configForEffectiveTests(t, effectiveProfile())
|
||||
cfg.PromptKit = tt.profileSource
|
||||
effective, err := cfg.Resolve(ResolveInput{PipelineID: "main", Catalog: effectiveCatalog(t)})
|
||||
if err != nil {
|
||||
t.Fatalf("Resolve() error = %v", err)
|
||||
}
|
||||
if effective.Config.PromptKit != cfg.PromptKit {
|
||||
t.Fatalf("effective PromptKit config = %#v, want %#v", effective.Config.PromptKit, cfg.PromptKit)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestEffectiveConfigOwnsPromptKitLocalBackend(t *testing.T) {
|
||||
cfg := configForEffectiveTests(t, effectiveProfile())
|
||||
cfg.PromptKit.LocalBackend = &PromptKitLocalBackendConfig{
|
||||
Endpoint: "http://localhost:8000/v1",
|
||||
ConcurrencyLimit: 2,
|
||||
}
|
||||
effective, err := cfg.Resolve(ResolveInput{PipelineID: "main", Catalog: effectiveCatalog(t)})
|
||||
if err != nil {
|
||||
t.Fatalf("Resolve() error = %v", err)
|
||||
}
|
||||
if effective.Config.PromptKit.LocalBackend == nil {
|
||||
t.Fatal("effective local backend = nil")
|
||||
}
|
||||
if effective.Config.PromptKit.LocalBackend == cfg.PromptKit.LocalBackend {
|
||||
t.Fatal("effective local backend aliases input config")
|
||||
}
|
||||
|
||||
cfg.PromptKit.LocalBackend.Endpoint = "http://changed-input.example/v1"
|
||||
if effective.Config.PromptKit.LocalBackend.Endpoint != "http://localhost:8000/v1" {
|
||||
t.Fatalf("input mutation changed effective config: %#v", effective.Config.PromptKit.LocalBackend)
|
||||
}
|
||||
effective.Config.PromptKit.LocalBackend.ConcurrencyLimit = 9
|
||||
if cfg.PromptKit.LocalBackend.ConcurrencyLimit != 2 {
|
||||
t.Fatalf("effective mutation changed input config: %#v", cfg.PromptKit.LocalBackend)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEffectiveConfigResolutionFailuresRetainContext(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
@@ -154,7 +204,7 @@ func TestEffectiveConfigResolutionFailuresRetainContext(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestEffectiveConfigLLMProfileOverrideChangesDigestWithoutOverridingValidators(t *testing.T) {
|
||||
func TestEffectiveConfigLLMProfileOverrideChangesDigestAndOverridesValidators(t *testing.T) {
|
||||
profile := effectiveProfile()
|
||||
profile.Chunk.LLMProfile = "chunk-profile"
|
||||
lane := profile.Artifacts["lane"]
|
||||
@@ -187,8 +237,27 @@ func TestEffectiveConfigLLMProfileOverrideChangesDigestWithoutOverridingValidato
|
||||
t.Fatalf("pipeline profile override was not applied: %#v", resolved)
|
||||
}
|
||||
validators := findEffectiveValidatorChain(resolved, pipeline.StageExtract, "lane")
|
||||
if len(validators.Validators) != 1 || validators.Validators[0].Binding.LLMProfile != "validator-profile" {
|
||||
t.Fatalf("validator profile was overridden: %#v", validators)
|
||||
if len(validators.Validators) != 1 || validators.Validators[0].Binding.LLMProfile != "override-profile" {
|
||||
t.Fatalf("validator profile = %#v, want runtime override", validators)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEffectiveConfigPipelineLLMProfileIsInheritedWithoutMutatingConfig(t *testing.T) {
|
||||
profile := effectiveProfile()
|
||||
profile.LLMProfile = " configured-profile "
|
||||
effective, err := resolveEffectiveProfile(t, profile, ResolveInput{})
|
||||
if err != nil {
|
||||
t.Fatalf("Resolve() error = %v", err)
|
||||
}
|
||||
if got := effective.Config.Pipelines["main"].LLMProfile; got != " configured-profile " {
|
||||
t.Fatalf("effective config pipeline llm profile = %q, want preserved programmatic value", got)
|
||||
}
|
||||
resolved := effective.ResolvedPipeline
|
||||
if got := resolved.Chunk.LLMProfile; got != "configured-profile" {
|
||||
t.Fatalf("resolved chunk profile = %q, want inherited profile", got)
|
||||
}
|
||||
if got := resolved.Steps[0].ArtifactLanes[0].Extract.LLMProfile; got != "configured-profile" {
|
||||
t.Fatalf("resolved extract profile = %q, want inherited profile", got)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -435,7 +504,7 @@ func effectiveCatalog(t *testing.T) pipeline.ModuleCatalog {
|
||||
if err := pipeline.RegisterArtifactCodec(catalog.ArtifactCodecs, effectiveCodec{}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := catalog.Inputs.RegisterWithSpec(pipeline.ModuleSpec{Key: "input", Stage: pipeline.StageInput, Provides: []string{"source"}}, func() (contracts.InputAdapter, error) {
|
||||
if err := catalog.Inputs.RegisterWithSpec(pipeline.ModuleSpec{Key: "input", Stage: pipeline.StageInput, ExecutionClass: contracts.ExecutionClassDeterministic, Provides: []string{"source"}}, func() (contracts.InputAdapter, error) {
|
||||
return effectiveInput{key: "input"}, nil
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
@@ -443,6 +512,7 @@ func effectiveCatalog(t *testing.T) pipeline.ModuleCatalog {
|
||||
chunkSpec := pipeline.ModuleSpec{
|
||||
Key: pipeline.DefaultChunkModule,
|
||||
Stage: pipeline.StageChunk,
|
||||
ExecutionClass: contracts.ExecutionClassLLMBacked,
|
||||
Requires: []string{"source"},
|
||||
Provides: []string{"chunk"},
|
||||
ReferenceSlots: []contracts.ReferenceSlot{{Name: "chunk-ref"}},
|
||||
@@ -453,32 +523,32 @@ func effectiveCatalog(t *testing.T) pipeline.ModuleCatalog {
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := catalog.Chunkers.RegisterBuilderWithSpec(pipeline.ModuleSpec{Key: "needs-capability", Stage: pipeline.StageChunk, Requires: []string{"missing"}}, chunkOptions, func(pipeline.BuildRequest) (contracts.Chunker, error) {
|
||||
if err := catalog.Chunkers.RegisterBuilderWithSpec(pipeline.ModuleSpec{Key: "needs-capability", Stage: pipeline.StageChunk, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"missing"}}, chunkOptions, func(pipeline.BuildRequest) (contracts.Chunker, error) {
|
||||
return effectiveChunker{key: "needs-capability"}, nil
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := pipeline.RegisterExtractor(catalog.Extractors, pipeline.ModuleSpec{Key: "extract", Stage: pipeline.StageExtract, ArtifactKind: effectiveArtifactKind, Requires: []string{"chunk"}, Provides: []string{"candidate"}}, func() (contracts.Extractor[effectiveArtifact], error) {
|
||||
if err := pipeline.RegisterExtractor(catalog.Extractors, pipeline.ModuleSpec{Key: "extract", Stage: pipeline.StageExtract, ExecutionClass: contracts.ExecutionClassLLMBacked, ArtifactKind: effectiveArtifactKind, Requires: []string{"chunk"}, Provides: []string{"candidate"}}, func() (contracts.Extractor[effectiveArtifact], error) {
|
||||
return effectiveExtractor{key: "extract"}, nil
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := pipeline.RegisterMerger(catalog.Mergers, pipeline.ModuleSpec{Key: pipeline.DefaultMergeModule, Stage: pipeline.StageMerge, ArtifactKind: effectiveArtifactKind, Requires: []string{"candidate"}, Provides: []string{"merged"}}, func() (contracts.Merger[effectiveArtifact], error) {
|
||||
if err := pipeline.RegisterMerger(catalog.Mergers, pipeline.ModuleSpec{Key: pipeline.DefaultMergeModule, Stage: pipeline.StageMerge, ExecutionClass: contracts.ExecutionClassLLMBacked, ArtifactKind: effectiveArtifactKind, Requires: []string{"candidate"}, Provides: []string{"merged"}}, func() (contracts.Merger[effectiveArtifact], error) {
|
||||
return effectiveMerger{key: pipeline.DefaultMergeModule}, nil
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := pipeline.RegisterMerger(catalog.Mergers, pipeline.ModuleSpec{Key: "other-merge", Stage: pipeline.StageMerge, ArtifactKind: "other-kind", Requires: []string{"candidate"}, Provides: []string{"merged"}}, func() (contracts.Merger[effectiveArtifact], error) {
|
||||
if err := pipeline.RegisterMerger(catalog.Mergers, pipeline.ModuleSpec{Key: "other-merge", Stage: pipeline.StageMerge, ExecutionClass: contracts.ExecutionClassDeterministic, ArtifactKind: "other-kind", Requires: []string{"candidate"}, Provides: []string{"merged"}}, func() (contracts.Merger[effectiveArtifact], error) {
|
||||
return effectiveMerger{key: "other-merge"}, nil
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := pipeline.RegisterNormalizer(catalog.Normalizers, pipeline.ModuleSpec{Key: pipeline.DefaultNormalizeModule, Stage: pipeline.StageNormalize, ArtifactKind: effectiveArtifactKind, Requires: []string{"merged"}, Provides: []string{"normalized"}}, func() (contracts.Normalizer[effectiveArtifact], error) {
|
||||
if err := pipeline.RegisterNormalizer(catalog.Normalizers, pipeline.ModuleSpec{Key: pipeline.DefaultNormalizeModule, Stage: pipeline.StageNormalize, ExecutionClass: contracts.ExecutionClassLLMBacked, ArtifactKind: effectiveArtifactKind, Requires: []string{"merged"}, Provides: []string{"normalized"}}, func() (contracts.Normalizer[effectiveArtifact], error) {
|
||||
return effectiveNormalizer{key: pipeline.DefaultNormalizeModule}, nil
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := catalog.Outputs.RegisterWithSpec(pipeline.ModuleSpec{Key: pipeline.DefaultOutputModule, Stage: pipeline.StageOutput, Requires: []string{"normalized"}}, func() (contracts.OutputEncoder, error) {
|
||||
if err := catalog.Outputs.RegisterWithSpec(pipeline.ModuleSpec{Key: pipeline.DefaultOutputModule, Stage: pipeline.StageOutput, ExecutionClass: contracts.ExecutionClassDeterministic, Requires: []string{"normalized"}}, func() (contracts.OutputEncoder, error) {
|
||||
return effectiveOutput{key: pipeline.DefaultOutputModule}, nil
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
|
||||
@@ -10,7 +10,7 @@ import (
|
||||
)
|
||||
|
||||
func TestPrecedenceFileValuesOverrideBuiltInDefaults(t *testing.T) {
|
||||
cfg := applyFileConfig(t, `version: 3
|
||||
cfg := applyFileConfig(t, `version: 4
|
||||
concurrency:
|
||||
total_llm: 4
|
||||
stage_workers:
|
||||
@@ -35,7 +35,7 @@ debug:
|
||||
}
|
||||
|
||||
func TestPrecedenceOperationalEnvironmentOverridesFileValues(t *testing.T) {
|
||||
cfg := applyFileConfig(t, `version: 3
|
||||
cfg := applyFileConfig(t, `version: 4
|
||||
concurrency:
|
||||
total_llm: 2
|
||||
stage_workers:
|
||||
@@ -81,21 +81,21 @@ func TestPrecedenceExtractWorkersFollowEffectiveConcurrencyUnlessExplicit(t *tes
|
||||
}{
|
||||
{
|
||||
name: "default follows environment total",
|
||||
file: "version: 3\n",
|
||||
file: "version: 4\n",
|
||||
env: map[string]string{"NOTARIUS_TOTAL_LLM_CONCURRENCY": "5"},
|
||||
wantTotal: 5,
|
||||
wantWorker: 5,
|
||||
},
|
||||
{
|
||||
name: "file worker is retained",
|
||||
file: "version: 3\nconcurrency:\n total_llm: 3\n stage_workers:\n extract: 2\n",
|
||||
file: "version: 4\nconcurrency:\n total_llm: 3\n stage_workers:\n extract: 2\n",
|
||||
env: map[string]string{"NOTARIUS_TOTAL_LLM_CONCURRENCY": "6"},
|
||||
wantTotal: 6,
|
||||
wantWorker: 2,
|
||||
},
|
||||
{
|
||||
name: "environment worker is retained",
|
||||
file: "version: 3\nconcurrency:\n total_llm: 2\n",
|
||||
file: "version: 4\nconcurrency:\n total_llm: 2\n",
|
||||
env: map[string]string{
|
||||
"NOTARIUS_TOTAL_LLM_CONCURRENCY": "6",
|
||||
"NOTARIUS_STAGE_WORKERS_EXTRACT": "4",
|
||||
@@ -118,7 +118,7 @@ func TestPrecedenceExtractWorkersFollowEffectiveConcurrencyUnlessExplicit(t *tes
|
||||
}
|
||||
|
||||
func TestPrecedenceEmptyFileCacheDirectoriesDeferPerUserResolution(t *testing.T) {
|
||||
cfg := applyFileConfig(t, `version: 3
|
||||
cfg := applyFileConfig(t, `version: 4
|
||||
cache:
|
||||
chunk_plans:
|
||||
directory: ""
|
||||
|
||||
@@ -14,7 +14,7 @@ import (
|
||||
|
||||
type FileConfig struct {
|
||||
Version int `yaml:"version"`
|
||||
Scriptorium *FileScriptoriumConfig `yaml:"scriptorium,omitempty"`
|
||||
PromptKit *FilePromptKitConfig `yaml:"promptkit,omitempty"`
|
||||
Pipelines map[string]FilePipelineProfile `yaml:"pipelines,omitempty"`
|
||||
Concurrency *FileConcurrencyConfig `yaml:"concurrency,omitempty"`
|
||||
Output *FileOutputConfig `yaml:"output,omitempty"`
|
||||
@@ -22,12 +22,19 @@ type FileConfig struct {
|
||||
Debug *FileDebugConfig `yaml:"debug,omitempty"`
|
||||
}
|
||||
|
||||
type FileScriptoriumConfig struct {
|
||||
type FilePromptKitConfig struct {
|
||||
ProfileDir *string `yaml:"profile_dir,omitempty"`
|
||||
ProfileFile *string `yaml:"profile_file,omitempty"`
|
||||
LocalBackend *FilePromptKitLocalBackendConfig `yaml:"local_backend,omitempty"`
|
||||
}
|
||||
|
||||
type FilePromptKitLocalBackendConfig struct {
|
||||
Endpoint *string `yaml:"endpoint,omitempty"`
|
||||
ConcurrencyLimit *int `yaml:"concurrency_limit,omitempty"`
|
||||
}
|
||||
|
||||
type FilePipelineProfile struct {
|
||||
LLMProfile *string `yaml:"llm_profile,omitempty"`
|
||||
Input fileModuleBinding `yaml:"input"`
|
||||
Chunk *fileModuleBinding `yaml:"chunk,omitempty"`
|
||||
Artifacts map[string]FileArtifactLaneProfile `yaml:"artifacts,omitempty"`
|
||||
@@ -36,13 +43,14 @@ type FilePipelineProfile struct {
|
||||
References map[string]fileReferenceSource `yaml:"references,omitempty"`
|
||||
artifactsSet bool `yaml:"-"`
|
||||
stepsSet bool `yaml:"-"`
|
||||
llmProfileSet bool `yaml:"-"`
|
||||
}
|
||||
|
||||
func (p *FilePipelineProfile) UnmarshalYAML(node *yaml.Node) error {
|
||||
type plainFilePipelineProfile FilePipelineProfile
|
||||
var decoded plainFilePipelineProfile
|
||||
seen, err := decodeKnownMapping(node, &decoded, map[string]struct{}{
|
||||
"input": {}, "chunk": {}, "artifacts": {}, "steps": {}, "output": {}, "references": {},
|
||||
"llm_profile": {}, "input": {}, "chunk": {}, "artifacts": {}, "steps": {}, "output": {}, "references": {},
|
||||
}, "pipeline profile")
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -50,6 +58,7 @@ func (p *FilePipelineProfile) UnmarshalYAML(node *yaml.Node) error {
|
||||
*p = FilePipelineProfile(decoded)
|
||||
_, p.artifactsSet = seen["artifacts"]
|
||||
_, p.stepsSet = seen["steps"]
|
||||
_, p.llmProfileSet = seen["llm_profile"]
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -251,6 +260,9 @@ func (b *fileModuleBinding) UnmarshalYAML(node *yaml.Node) error {
|
||||
return err
|
||||
}
|
||||
b.LLMProfile = strings.TrimSpace(llmProfile)
|
||||
if b.LLMProfile == "" {
|
||||
return fmt.Errorf("llm_profile must not be empty when set")
|
||||
}
|
||||
case "retries":
|
||||
var retries int
|
||||
if err := valueNode.Decode(&retries); err != nil {
|
||||
@@ -325,6 +337,9 @@ func ParseFileConfigYAML(data []byte) (FileConfig, error) {
|
||||
if header.Version == 2 {
|
||||
return FileConfig{}, fmt.Errorf("config version 2 is no longer supported; migrate the file using the version 2-to-3 migration in docs/config.md")
|
||||
}
|
||||
if header.Version == 3 {
|
||||
return FileConfig{}, fmt.Errorf("config version 3 is no longer supported; change \"version: 3\" to \"version: 4\" and rename \"scriptorium:\" to \"promptkit:\"")
|
||||
}
|
||||
if header.Version != SupportedFileConfigVersion {
|
||||
return FileConfig{}, fmt.Errorf("unsupported config version %d (supported version is %d)", header.Version, SupportedFileConfigVersion)
|
||||
}
|
||||
@@ -450,25 +465,46 @@ func (c *Config) applyFileConfigWithLookup(fileCfg FileConfig, lookup func(strin
|
||||
}
|
||||
}
|
||||
|
||||
if fileCfg.Scriptorium != nil {
|
||||
if fileCfg.Scriptorium.ProfileDir != nil {
|
||||
value := strings.TrimSpace(*fileCfg.Scriptorium.ProfileDir)
|
||||
if fileCfg.PromptKit != nil {
|
||||
if fileCfg.PromptKit.ProfileDir != nil {
|
||||
value := strings.TrimSpace(*fileCfg.PromptKit.ProfileDir)
|
||||
if value == "" {
|
||||
return fmt.Errorf("scriptorium.profile_dir must not be empty when set")
|
||||
return fmt.Errorf("promptkit.profile_dir must not be empty when set")
|
||||
}
|
||||
c.Scriptorium.ProfileDir = value
|
||||
c.PromptKit.ProfileDir = value
|
||||
}
|
||||
if fileCfg.Scriptorium.ProfileFile != nil {
|
||||
value := strings.TrimSpace(*fileCfg.Scriptorium.ProfileFile)
|
||||
if fileCfg.PromptKit.ProfileFile != nil {
|
||||
value := strings.TrimSpace(*fileCfg.PromptKit.ProfileFile)
|
||||
if value == "" {
|
||||
return fmt.Errorf("scriptorium.profile_file must not be empty when set")
|
||||
return fmt.Errorf("promptkit.profile_file must not be empty when set")
|
||||
}
|
||||
c.Scriptorium.ProfileFile = value
|
||||
c.PromptKit.ProfileFile = value
|
||||
}
|
||||
if fileCfg.PromptKit.LocalBackend != nil {
|
||||
if fileCfg.PromptKit.LocalBackend.Endpoint == nil {
|
||||
return fmt.Errorf("promptkit.local_backend.endpoint must not be empty when set")
|
||||
}
|
||||
endpoint := strings.TrimSpace(*fileCfg.PromptKit.LocalBackend.Endpoint)
|
||||
if endpoint == "" {
|
||||
return fmt.Errorf("promptkit.local_backend.endpoint must not be empty when set")
|
||||
}
|
||||
localBackend := PromptKitLocalBackendConfig{Endpoint: endpoint}
|
||||
if fileCfg.PromptKit.LocalBackend.ConcurrencyLimit != nil {
|
||||
localBackend.ConcurrencyLimit = *fileCfg.PromptKit.LocalBackend.ConcurrencyLimit
|
||||
}
|
||||
c.PromptKit.LocalBackend = &localBackend
|
||||
}
|
||||
}
|
||||
|
||||
for _, pipelineID := range pipelineIDs {
|
||||
filePipeline := fileCfg.Pipelines[rawPipelineIDs[pipelineID]]
|
||||
llmProfile := ""
|
||||
if filePipeline.llmProfileSet || filePipeline.LLMProfile != nil {
|
||||
if filePipeline.LLMProfile == nil || strings.TrimSpace(*filePipeline.LLMProfile) == "" {
|
||||
return fmt.Errorf("pipeline %q llm_profile must not be empty when set", pipelineID)
|
||||
}
|
||||
llmProfile = strings.TrimSpace(*filePipeline.LLMProfile)
|
||||
}
|
||||
hasSteps := filePipeline.stepsSet || filePipeline.Steps != nil
|
||||
laneIDs, rawLaneIDs, err := normalizedMapKeys(filePipeline.Artifacts, fmt.Sprintf("pipeline %q artifact lane id", pipelineID))
|
||||
if err != nil {
|
||||
@@ -476,6 +512,7 @@ func (c *Config) applyFileConfigWithLookup(fileCfg FileConfig, lookup func(strin
|
||||
}
|
||||
profile := pipeline.PipelineProfile{
|
||||
ID: pipelineID,
|
||||
LLMProfile: llmProfile,
|
||||
Input: filePipeline.Input.toPipelineBinding(),
|
||||
Artifacts: make(map[string]pipeline.ArtifactLaneProfile, len(filePipeline.Artifacts)),
|
||||
References: fileReferenceSourcesToPipeline(filePipeline.References),
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
package config
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"reflect"
|
||||
@@ -34,8 +36,8 @@ func TestDefaultReturnsDocumentedValuesAndIndependentMaps(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestFileConfigMinimalVersion3AppliesOverDefaults(t *testing.T) {
|
||||
file := parseFileConfig(t, "version: 3\n")
|
||||
func TestFileConfigMinimalVersion4AppliesOverDefaults(t *testing.T) {
|
||||
file := parseFileConfig(t, "version: 4\n")
|
||||
cfg := Default()
|
||||
if err := cfg.ApplyFileConfig(file); err != nil {
|
||||
t.Fatal(err)
|
||||
@@ -48,6 +50,259 @@ func TestFileConfigMinimalVersion3AppliesOverDefaults(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestFilePipelineLLMProfileIsPresenceAwareAndDetached(t *testing.T) {
|
||||
const pipelineYAML = `version: 4
|
||||
pipelines:
|
||||
main:
|
||||
%s
|
||||
input: input
|
||||
artifacts:
|
||||
lane:
|
||||
extract: extract
|
||||
`
|
||||
|
||||
t.Run("omitted", func(t *testing.T) {
|
||||
file := parseFileConfig(t, fmt.Sprintf(pipelineYAML, ""))
|
||||
if file.Pipelines["main"].LLMProfile != nil || file.Pipelines["main"].llmProfileSet {
|
||||
t.Fatalf("parsed pipeline profile = %#v, want omitted llm profile", file.Pipelines["main"])
|
||||
}
|
||||
cfg := Default()
|
||||
if err := cfg.ApplyFileConfig(file); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := cfg.Pipelines["main"].LLMProfile; got != "" {
|
||||
t.Fatalf("pipeline llm profile = %q, want empty", got)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("trimmed and detached", func(t *testing.T) {
|
||||
file := parseFileConfig(t, fmt.Sprintf(pipelineYAML, "llm_profile: ' configured-profile '"))
|
||||
cfg := Default()
|
||||
if err := cfg.ApplyFileConfig(file); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := cfg.Pipelines["main"].LLMProfile; got != "configured-profile" {
|
||||
t.Fatalf("pipeline llm profile = %q, want trimmed value", got)
|
||||
}
|
||||
*file.Pipelines["main"].LLMProfile = "changed-profile"
|
||||
if got := cfg.Pipelines["main"].LLMProfile; got != "configured-profile" {
|
||||
t.Fatalf("effective config aliases parsed file: %q", got)
|
||||
}
|
||||
if got := cloneConfig(cfg).Pipelines["main"].LLMProfile; got != "configured-profile" {
|
||||
t.Fatalf("cloned pipeline llm profile = %q", got)
|
||||
}
|
||||
data, err := json.Marshal(cfg)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var roundTripped Config
|
||||
if err := json.Unmarshal(data, &roundTripped); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := roundTripped.Pipelines["main"].LLMProfile; got != "configured-profile" {
|
||||
t.Fatalf("round-tripped pipeline llm profile = %q", got)
|
||||
}
|
||||
})
|
||||
|
||||
for _, value := range []string{"''", "' '", "null"} {
|
||||
t.Run("explicit empty "+value, func(t *testing.T) {
|
||||
file := parseFileConfig(t, fmt.Sprintf(pipelineYAML, "llm_profile: "+value))
|
||||
cfg := Default()
|
||||
err := cfg.ApplyFileConfig(file)
|
||||
if err == nil || !strings.Contains(err.Error(), `pipeline "main" llm_profile must not be empty`) {
|
||||
t.Fatalf("ApplyFileConfig() error = %v, want explicit-empty rejection", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestFileModuleBindingRejectsExplicitEmptyLLMProfile(t *testing.T) {
|
||||
const configYAML = `version: 4
|
||||
pipelines:
|
||||
main:
|
||||
input:
|
||||
module: input
|
||||
llm_profile: %s
|
||||
artifacts:
|
||||
lane:
|
||||
extract: extract
|
||||
`
|
||||
for _, value := range []string{"''", "' '", "null"} {
|
||||
t.Run(value, func(t *testing.T) {
|
||||
_, err := ParseFileConfigYAML([]byte(fmt.Sprintf(configYAML, value)))
|
||||
if err == nil || !strings.Contains(err.Error(), "llm_profile must not be empty when set") {
|
||||
t.Fatalf("ParseFileConfigYAML() error = %v, want explicit-empty binding profile rejection", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestFilePromptKitProfileSourcesSurviveConfigBoundaries(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
yaml string
|
||||
want PromptKitConfig
|
||||
}{
|
||||
{
|
||||
name: "profile directory",
|
||||
yaml: "version: 4\npromptkit:\n profile_dir: ' ./profiles '\n",
|
||||
want: PromptKitConfig{ProfileDir: "./profiles"},
|
||||
},
|
||||
{
|
||||
name: "profile file",
|
||||
yaml: "version: 4\npromptkit:\n profile_file: ' ./profiles.yml '\n",
|
||||
want: PromptKitConfig{ProfileFile: "./profiles.yml"},
|
||||
},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
cfg := applyFileConfig(t, tt.yaml)
|
||||
if cfg.PromptKit != tt.want {
|
||||
t.Fatalf("PromptKit config = %#v, want %#v", cfg.PromptKit, tt.want)
|
||||
}
|
||||
if got := cloneConfig(cfg).PromptKit; got != tt.want {
|
||||
t.Fatalf("cloned PromptKit config = %#v, want %#v", got, tt.want)
|
||||
}
|
||||
if got := cfg.Redacted().PromptKit; got != tt.want {
|
||||
t.Fatalf("redacted PromptKit config = %#v, want %#v", got, tt.want)
|
||||
}
|
||||
|
||||
data, err := json.Marshal(cfg)
|
||||
if err != nil {
|
||||
t.Fatalf("json.Marshal() error = %v", err)
|
||||
}
|
||||
var payload map[string]json.RawMessage
|
||||
if err := json.Unmarshal(data, &payload); err != nil {
|
||||
t.Fatalf("json.Unmarshal() error = %v", err)
|
||||
}
|
||||
if _, ok := payload["promptkit"]; !ok {
|
||||
t.Fatalf("runtime JSON keys = %v, want promptkit", payload)
|
||||
}
|
||||
if _, ok := payload["scriptorium"]; ok {
|
||||
t.Fatalf("runtime JSON keys = %v, must not contain removed section", payload)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestFilePromptKitLocalBackendSurvivesConfigBoundaries(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
concurrencyYAML string
|
||||
wantConcurrency int
|
||||
}{
|
||||
{name: "omitted concurrency defaults to zero"},
|
||||
{name: "positive concurrency is preserved", concurrencyYAML: " concurrency_limit: 2\n", wantConcurrency: 2},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
file := parseFileConfig(t, "version: 4\npromptkit:\n local_backend:\n endpoint: ' http://localhost:8000/v1 '\n"+tt.concurrencyYAML)
|
||||
cfg := Default()
|
||||
if err := cfg.ApplyFileConfig(file); err != nil {
|
||||
t.Fatalf("ApplyFileConfig() error = %v", err)
|
||||
}
|
||||
want := PromptKitLocalBackendConfig{
|
||||
Endpoint: "http://localhost:8000/v1",
|
||||
ConcurrencyLimit: tt.wantConcurrency,
|
||||
}
|
||||
if cfg.PromptKit.LocalBackend == nil || *cfg.PromptKit.LocalBackend != want {
|
||||
t.Fatalf("local backend config = %#v, want %#v", cfg.PromptKit.LocalBackend, want)
|
||||
}
|
||||
|
||||
*file.PromptKit.LocalBackend.Endpoint = "http://changed.example/v1"
|
||||
if file.PromptKit.LocalBackend.ConcurrencyLimit != nil {
|
||||
*file.PromptKit.LocalBackend.ConcurrencyLimit = 99
|
||||
}
|
||||
if *cfg.PromptKit.LocalBackend != want {
|
||||
t.Fatalf("effective config aliases parsed file model: %#v", cfg.PromptKit.LocalBackend)
|
||||
}
|
||||
|
||||
cloned := cloneConfig(cfg)
|
||||
if cloned.PromptKit.LocalBackend == cfg.PromptKit.LocalBackend || *cloned.PromptKit.LocalBackend != want {
|
||||
t.Fatalf("cloned local backend = %#v, want detached %#v", cloned.PromptKit.LocalBackend, want)
|
||||
}
|
||||
cloned.PromptKit.LocalBackend.Endpoint = "http://clone.example/v1"
|
||||
if *cfg.PromptKit.LocalBackend != want {
|
||||
t.Fatalf("mutating clone changed source config: %#v", cfg.PromptKit.LocalBackend)
|
||||
}
|
||||
|
||||
redacted := cfg.Redacted()
|
||||
if redacted.PromptKit.LocalBackend == cfg.PromptKit.LocalBackend || *redacted.PromptKit.LocalBackend != want {
|
||||
t.Fatalf("redacted local backend = %#v, want detached %#v", redacted.PromptKit.LocalBackend, want)
|
||||
}
|
||||
|
||||
data, err := json.Marshal(cfg)
|
||||
if err != nil {
|
||||
t.Fatalf("json.Marshal() error = %v", err)
|
||||
}
|
||||
var payload struct {
|
||||
PromptKit map[string]json.RawMessage `json:"promptkit"`
|
||||
}
|
||||
if err := json.Unmarshal(data, &payload); err != nil {
|
||||
t.Fatalf("json.Unmarshal() error = %v", err)
|
||||
}
|
||||
localJSON, ok := payload.PromptKit["local_backend"]
|
||||
if !ok {
|
||||
t.Fatalf("runtime PromptKit JSON keys = %v, want local_backend", payload.PromptKit)
|
||||
}
|
||||
var localPayload map[string]json.RawMessage
|
||||
if err := json.Unmarshal(localJSON, &localPayload); err != nil {
|
||||
t.Fatalf("unmarshal local_backend JSON: %v", err)
|
||||
}
|
||||
if _, ok := localPayload["endpoint"]; !ok {
|
||||
t.Fatalf("runtime local_backend JSON keys = %v, want endpoint", localPayload)
|
||||
}
|
||||
if _, ok := localPayload["concurrency_limit"]; !ok {
|
||||
t.Fatalf("runtime local_backend JSON keys = %v, want concurrency_limit", localPayload)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestFilePromptKitLocalBackendRequiresEndpoint(t *testing.T) {
|
||||
for _, tt := range []struct {
|
||||
name string
|
||||
yaml string
|
||||
}{
|
||||
{name: "missing", yaml: "version: 4\npromptkit:\n local_backend: {}\n"},
|
||||
{name: "empty", yaml: "version: 4\npromptkit:\n local_backend:\n endpoint: ''\n"},
|
||||
{name: "blank", yaml: "version: 4\npromptkit:\n local_backend:\n endpoint: ' '\n"},
|
||||
} {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
file := parseFileConfig(t, tt.yaml)
|
||||
cfg := Default()
|
||||
err := cfg.ApplyFileConfig(file)
|
||||
if err == nil || !strings.Contains(err.Error(), "promptkit.local_backend.endpoint") {
|
||||
t.Fatalf("ApplyFileConfig() error = %v, want endpoint field context", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestFilePromptKitExplicitEmptyProfileSourcesAreRejected(t *testing.T) {
|
||||
for _, field := range []string{"profile_dir", "profile_file"} {
|
||||
t.Run(field, func(t *testing.T) {
|
||||
file := parseFileConfig(t, "version: 4\npromptkit:\n "+field+": ''\n")
|
||||
cfg := Default()
|
||||
err := cfg.ApplyFileConfig(file)
|
||||
if err == nil || !strings.Contains(err.Error(), "promptkit."+field+" must not be empty") {
|
||||
t.Fatalf("ApplyFileConfig() error = %v, want explicit-empty rejection", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestFilePromptKitProfileSourcesRemainMutuallyExclusive(t *testing.T) {
|
||||
cfg := applyFileConfig(t, `version: 4
|
||||
promptkit:
|
||||
profile_dir: ./profiles
|
||||
profile_file: ./profiles.yml
|
||||
`)
|
||||
if err := cfg.Validate(); err == nil || !strings.Contains(err.Error(), "promptkit profile_dir and profile_file are mutually exclusive") {
|
||||
t.Fatalf("Validate() error = %v, want mutually exclusive profile sources", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFileConfigMissingVersionIsReportedBeforeFieldDecoding(t *testing.T) {
|
||||
_, err := ParseFileConfigYAML([]byte("workspace:\n directory: /tmp/old\n"))
|
||||
if err == nil || !strings.Contains(err.Error(), "config version is required") {
|
||||
@@ -55,6 +310,15 @@ func TestFileConfigMissingVersionIsReportedBeforeFieldDecoding(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestFileConfigVersion3ReportsPromptKitMigration(t *testing.T) {
|
||||
_, err := ParseFileConfigYAML([]byte("version: 3\nscriptorium:\n profile_dir: ./profiles\n"))
|
||||
if err == nil ||
|
||||
!strings.Contains(err.Error(), `change "version: 3" to "version: 4"`) ||
|
||||
!strings.Contains(err.Error(), `rename "scriptorium:" to "promptkit:"`) {
|
||||
t.Fatalf("version 3 error = %v, want actionable version and section migration", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFileConfigRejectsUnknownCurrentAndRemovedFields(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
@@ -63,14 +327,19 @@ func TestFileConfigRejectsUnknownCurrentAndRemovedFields(t *testing.T) {
|
||||
}{
|
||||
{
|
||||
name: "removed diagnostics",
|
||||
yaml: "version: 3\ndiagnostics: {}\n",
|
||||
yaml: "version: 4\ndiagnostics: {}\n",
|
||||
want: "field diagnostics not found",
|
||||
},
|
||||
{
|
||||
name: "removed llm profiles",
|
||||
yaml: "version: 3\nllm_profiles: {}\n",
|
||||
yaml: "version: 4\nllm_profiles: {}\n",
|
||||
want: "field llm_profiles not found",
|
||||
},
|
||||
{
|
||||
name: "removed scriptorium section",
|
||||
yaml: "version: 4\nscriptorium: {}\n",
|
||||
want: "field scriptorium not found",
|
||||
},
|
||||
{
|
||||
name: "version 2 migration",
|
||||
yaml: "version: 2\nworkspace:\n directory: /tmp/old\n",
|
||||
@@ -78,29 +347,34 @@ func TestFileConfigRejectsUnknownCurrentAndRemovedFields(t *testing.T) {
|
||||
},
|
||||
{
|
||||
name: "pipeline field",
|
||||
yaml: "version: 3\npipelines:\n main:\n unknown: true\n",
|
||||
yaml: "version: 4\npipelines:\n main:\n unknown: true\n",
|
||||
want: "field unknown not found",
|
||||
},
|
||||
{
|
||||
name: "lane field",
|
||||
yaml: "version: 3\npipelines:\n main:\n artifacts:\n spells:\n unknown: true\n",
|
||||
yaml: "version: 4\npipelines:\n main:\n artifacts:\n spells:\n unknown: true\n",
|
||||
want: "field unknown not found",
|
||||
},
|
||||
{
|
||||
name: "module binding field",
|
||||
yaml: "version: 3\npipelines:\n main:\n input:\n module: seriatim\n unknown: true\n",
|
||||
yaml: "version: 4\npipelines:\n main:\n input:\n module: seriatim\n unknown: true\n",
|
||||
want: "field unknown not found in module binding",
|
||||
},
|
||||
{
|
||||
name: "checkpoint field",
|
||||
yaml: "version: 3\ncache:\n checkpoints:\n unknown: true\n",
|
||||
yaml: "version: 4\ncache:\n checkpoints:\n unknown: true\n",
|
||||
want: "field unknown not found",
|
||||
},
|
||||
{
|
||||
name: "checkpoint enabled type",
|
||||
yaml: "version: 3\ncache:\n checkpoints:\n enabled: definitely\n",
|
||||
yaml: "version: 4\ncache:\n checkpoints:\n enabled: definitely\n",
|
||||
want: "cannot unmarshal",
|
||||
},
|
||||
{
|
||||
name: "local backend field",
|
||||
yaml: "version: 4\npromptkit:\n local_backend:\n endpoint: http://localhost:8000/v1\n unknown: true\n",
|
||||
want: "field unknown not found",
|
||||
},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
@@ -113,7 +387,7 @@ func TestFileConfigRejectsUnknownCurrentAndRemovedFields(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestFileConfigModuleBindingsPreserveFormsAndValidatorPresence(t *testing.T) {
|
||||
cfg := applyFileConfig(t, `version: 3
|
||||
cfg := applyFileConfig(t, `version: 4
|
||||
pipelines:
|
||||
main:
|
||||
input: seriatim
|
||||
@@ -160,7 +434,7 @@ pipelines:
|
||||
}
|
||||
|
||||
func TestFileConfigReferencePrecedenceIsRetained(t *testing.T) {
|
||||
cfg := applyFileConfig(t, `version: 3
|
||||
cfg := applyFileConfig(t, `version: 4
|
||||
pipelines:
|
||||
main:
|
||||
input: seriatim
|
||||
@@ -224,7 +498,7 @@ pipelines:
|
||||
}
|
||||
|
||||
func TestFileConfigStageLocalValidatorsPreserveOrderAndFields(t *testing.T) {
|
||||
cfg := applyFileConfig(t, `version: 3
|
||||
cfg := applyFileConfig(t, `version: 4
|
||||
pipelines:
|
||||
main:
|
||||
input: seriatim
|
||||
@@ -277,8 +551,8 @@ pipelines:
|
||||
}
|
||||
|
||||
func TestFileConfigStateSectionsApplyIndependently(t *testing.T) {
|
||||
cfg := applyFileConfig(t, `version: 3
|
||||
scriptorium:
|
||||
cfg := applyFileConfig(t, `version: 4
|
||||
promptkit:
|
||||
profile_dir: ./profiles
|
||||
concurrency:
|
||||
total_llm: 7
|
||||
@@ -294,15 +568,15 @@ cache:
|
||||
debug:
|
||||
directory: ./debug
|
||||
`)
|
||||
if cfg.Scriptorium.ProfileDir != "./profiles" || cfg.Scriptorium.ProfileFile != "" {
|
||||
t.Fatalf("scriptorium = %#v", cfg.Scriptorium)
|
||||
if cfg.PromptKit.ProfileDir != "./profiles" || cfg.PromptKit.ProfileFile != "" {
|
||||
t.Fatalf("promptkit = %#v", cfg.PromptKit)
|
||||
}
|
||||
if cfg.Concurrency.TotalLLM != 7 || cfg.Concurrency.StageWorkers["extract"] != 7 {
|
||||
t.Fatalf("concurrency = %#v", cfg.Concurrency)
|
||||
}
|
||||
if cfg.Output.Directory != "./output" || cfg.Cache.ChunkPlans.Directory != "plans" || cfg.Cache.ChunkPlans.Mode != pipeline.ChunkCacheBypass ||
|
||||
!cfg.Cache.Checkpoints.Enabled || cfg.Cache.Checkpoints.Directory != "checkpoints" || cfg.Debug.Directory != "./debug" {
|
||||
t.Fatalf("state sections = %#v, %#v, %#v, %#v", cfg.Output, cfg.Cache, cfg.Debug, cfg.Scriptorium)
|
||||
t.Fatalf("state sections = %#v, %#v, %#v, %#v", cfg.Output, cfg.Cache, cfg.Debug, cfg.PromptKit)
|
||||
}
|
||||
if cfg.Output.Directory == cfg.Cache.ChunkPlans.Directory || cfg.Cache.ChunkPlans.Directory == cfg.Cache.Checkpoints.Directory || cfg.Cache.Checkpoints.Directory == cfg.Debug.Directory {
|
||||
t.Fatal("state roots were coupled")
|
||||
@@ -310,11 +584,11 @@ debug:
|
||||
}
|
||||
|
||||
func TestFileConfigCheckpointEnabledCanBeExplicitlyDisabled(t *testing.T) {
|
||||
cfg := applyFileConfig(t, "version: 3\ncache:\n checkpoints:\n enabled: true\n")
|
||||
cfg := applyFileConfig(t, "version: 4\ncache:\n checkpoints:\n enabled: true\n")
|
||||
if !cfg.Cache.Checkpoints.Enabled || !cloneConfig(cfg).Cache.Checkpoints.Enabled {
|
||||
t.Fatalf("enabled checkpoint config was not retained: %#v", cfg.Cache.Checkpoints)
|
||||
}
|
||||
file := parseFileConfig(t, "version: 3\ncache:\n checkpoints:\n enabled: false\n")
|
||||
file := parseFileConfig(t, "version: 4\ncache:\n checkpoints:\n enabled: false\n")
|
||||
if err := cfg.ApplyFileConfig(file); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
@@ -331,17 +605,17 @@ func TestFileConfigRejectsTrimmedKeyCollisions(t *testing.T) {
|
||||
}{
|
||||
{
|
||||
name: "pipeline ids",
|
||||
yaml: "version: 3\npipelines:\n main: {}\n ' main ': {}\n",
|
||||
yaml: "version: 4\npipelines:\n main: {}\n ' main ': {}\n",
|
||||
want: "pipeline id \"main\" is duplicated after trimming",
|
||||
},
|
||||
{
|
||||
name: "lane ids",
|
||||
yaml: "version: 3\npipelines:\n main:\n artifacts:\n spells: {}\n ' spells ': {}\n",
|
||||
yaml: "version: 4\npipelines:\n main:\n artifacts:\n spells: {}\n ' spells ': {}\n",
|
||||
want: "artifact lane id \"spells\" is duplicated after trimming",
|
||||
},
|
||||
{
|
||||
name: "reference slots",
|
||||
yaml: "version: 3\npipelines:\n main:\n references:\n slot: ./one.txt\n ' slot ': ./two.txt\n",
|
||||
yaml: "version: 4\npipelines:\n main:\n references:\n slot: ./one.txt\n ' slot ': ./two.txt\n",
|
||||
want: "reference slot \"slot\" is duplicated after trimming",
|
||||
},
|
||||
}
|
||||
@@ -358,7 +632,7 @@ func TestFileConfigRejectsTrimmedKeyCollisions(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestFileConfigParsesOrderedStepsAndReferenceSources(t *testing.T) {
|
||||
file := parseFileConfig(t, `version: 3
|
||||
file := parseFileConfig(t, `version: 4
|
||||
pipelines:
|
||||
session:
|
||||
input: seriatim
|
||||
@@ -397,7 +671,7 @@ func TestFileConfigRejectsAmbiguousReferenceSourceForms(t *testing.T) {
|
||||
"artifact: {step: 1, lane: b}",
|
||||
"1",
|
||||
} {
|
||||
_, err := ParseFileConfigYAML([]byte("version: 3\npipelines:\n p:\n input: text\n references:\n slot: " + source + "\n"))
|
||||
_, err := ParseFileConfigYAML([]byte("version: 4\npipelines:\n p:\n input: text\n references:\n slot: " + source + "\n"))
|
||||
if err == nil {
|
||||
t.Fatalf("ParseFileConfigYAML(%q) error = nil", source)
|
||||
}
|
||||
@@ -412,12 +686,12 @@ func TestFileConfigRejectsEmptyAndAmbiguousPipelineShapes(t *testing.T) {
|
||||
}{
|
||||
{
|
||||
name: "empty steps",
|
||||
yaml: "version: 3\npipelines:\n p:\n input: text\n steps: []\n",
|
||||
yaml: "version: 4\npipelines:\n p:\n input: text\n steps: []\n",
|
||||
want: "at least one ordered step",
|
||||
},
|
||||
{
|
||||
name: "both forms",
|
||||
yaml: "version: 3\npipelines:\n p:\n input: text\n artifacts: {}\n steps: []\n",
|
||||
yaml: "version: 4\npipelines:\n p:\n input: text\n artifacts: {}\n steps: []\n",
|
||||
want: "both artifacts and steps",
|
||||
},
|
||||
}
|
||||
|
||||
@@ -22,7 +22,9 @@ func TestRedactedResolvedPipelinePayloadRedactsEveryBinding(t *testing.T) {
|
||||
ID: "redaction-test",
|
||||
Digest: "sha256:safe-digest",
|
||||
Input: bindings["input"],
|
||||
InputExecutionClass: contracts.ExecutionClassDeterministic,
|
||||
Chunk: bindings["chunk"],
|
||||
ChunkExecutionClass: contracts.ExecutionClassLLMBacked,
|
||||
ChunkReferences: redactionTestReferenceTarget(pipeline.StageChunk, "", "chunk-reference-content"),
|
||||
Steps: []pipeline.ResolvedPipelineStep{{
|
||||
ID: "default",
|
||||
@@ -30,8 +32,11 @@ func TestRedactedResolvedPipelinePayloadRedactsEveryBinding(t *testing.T) {
|
||||
ID: "safe-lane",
|
||||
ArtifactKind: "safe/artifact",
|
||||
Extract: bindings["extract"],
|
||||
ExtractExecutionClass: contracts.ExecutionClassLLMBacked,
|
||||
Merge: bindings["merge"],
|
||||
MergeExecutionClass: contracts.ExecutionClassDeterministic,
|
||||
Normalize: bindings["normalize"],
|
||||
NormalizeExecutionClass: contracts.ExecutionClassLLMBacked,
|
||||
Validators: []pipeline.ModuleBinding{bindings["lane-validator"]},
|
||||
ExtractReferences: redactionTestReferenceTarget(pipeline.StageExtract, "safe-lane", "extract-reference-content"),
|
||||
MergeReferences: redactionTestReferenceTarget(pipeline.StageMerge, "safe-lane", "merge-reference-content"),
|
||||
@@ -50,6 +55,7 @@ func TestRedactedResolvedPipelinePayloadRedactsEveryBinding(t *testing.T) {
|
||||
}},
|
||||
}},
|
||||
Output: bindings["output"],
|
||||
OutputExecutionClass: contracts.ExecutionClassDeterministic,
|
||||
}
|
||||
effective := EffectiveConfig{
|
||||
Config: Config{Pipelines: map[string]pipeline.PipelineProfile{
|
||||
@@ -88,6 +94,11 @@ func TestRedactedResolvedPipelinePayloadRedactsEveryBinding(t *testing.T) {
|
||||
t.Fatalf("resolved pipeline summary does not retain %q: %s", safe, text)
|
||||
}
|
||||
}
|
||||
for _, executionClass := range []string{"input_execution_class\":\"deterministic", "chunk_execution_class\":\"llm_backed", "extract_execution_class\":\"llm_backed", "merge_execution_class\":\"deterministic", "normalize_execution_class\":\"llm_backed", "output_execution_class\":\"deterministic"} {
|
||||
if !strings.Contains(text, executionClass) {
|
||||
t.Fatalf("resolved pipeline summary does not retain %q: %s", executionClass, text)
|
||||
}
|
||||
}
|
||||
|
||||
payload.Input.Options["safe"] = "mutated"
|
||||
nested := payload.Input.Options["nested"].([]any)[0].([]any)[0].(map[string]any)
|
||||
|
||||
@@ -2,6 +2,7 @@ package config
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"net/url"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
@@ -10,7 +11,7 @@ import (
|
||||
|
||||
func (c Config) Validate() error {
|
||||
c.Concurrency.recomputeStageWorkerDefaults()
|
||||
if err := validateScriptorium(c.Scriptorium); err != nil {
|
||||
if err := validatePromptKit(c.PromptKit); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := validateStateSurfaces(c); err != nil {
|
||||
@@ -49,9 +50,30 @@ func validateStageWorkers(cfg ConcurrencyConfig) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
func validateScriptorium(cfg ScriptoriumConfig) error {
|
||||
func validatePromptKit(cfg PromptKitConfig) error {
|
||||
if strings.TrimSpace(cfg.ProfileDir) != "" && strings.TrimSpace(cfg.ProfileFile) != "" {
|
||||
return fmt.Errorf("scriptorium profile_dir and profile_file are mutually exclusive")
|
||||
return fmt.Errorf("promptkit profile_dir and profile_file are mutually exclusive")
|
||||
}
|
||||
if cfg.LocalBackend == nil {
|
||||
return nil
|
||||
}
|
||||
endpoint := strings.TrimSpace(cfg.LocalBackend.Endpoint)
|
||||
if endpoint == "" {
|
||||
return fmt.Errorf("promptkit.local_backend.endpoint must not be empty when set")
|
||||
}
|
||||
parsed, err := url.Parse(endpoint)
|
||||
if err != nil ||
|
||||
(!strings.EqualFold(parsed.Scheme, "http") && !strings.EqualFold(parsed.Scheme, "https")) ||
|
||||
!parsed.IsAbs() ||
|
||||
parsed.Hostname() == "" ||
|
||||
parsed.User != nil ||
|
||||
parsed.RawQuery != "" ||
|
||||
parsed.ForceQuery ||
|
||||
strings.Contains(endpoint, "#") {
|
||||
return fmt.Errorf("promptkit.local_backend.endpoint must be an absolute HTTP or HTTPS URL with a host and no user information, query, or fragment")
|
||||
}
|
||||
if cfg.LocalBackend.ConcurrencyLimit < 0 {
|
||||
return fmt.Errorf("promptkit.local_backend.concurrency_limit must not be negative")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -87,10 +87,77 @@ func TestValidateConcurrencyRules(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidateScriptoriumSourcesAreMutuallyExclusive(t *testing.T) {
|
||||
func TestValidatePromptKitSourcesAreMutuallyExclusive(t *testing.T) {
|
||||
cfg := Default()
|
||||
cfg.Scriptorium = ScriptoriumConfig{ProfileDir: "./profiles", ProfileFile: "./profile.yml"}
|
||||
assertValidationContains(t, cfg, "scriptorium profile_dir and profile_file are mutually exclusive")
|
||||
cfg.PromptKit = PromptKitConfig{ProfileDir: "./profiles", ProfileFile: "./profile.yml"}
|
||||
assertValidationContains(t, cfg, "promptkit profile_dir and profile_file are mutually exclusive")
|
||||
}
|
||||
|
||||
func TestValidatePromptKitLocalBackendEndpoints(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
endpoint string
|
||||
profileSource PromptKitConfig
|
||||
}{
|
||||
{
|
||||
name: "HTTP endpoint with path and profile directory",
|
||||
endpoint: "http://localhost:8000/v1",
|
||||
profileSource: PromptKitConfig{ProfileDir: "./profiles"},
|
||||
},
|
||||
{
|
||||
name: "case-insensitive HTTPS endpoint and profile file",
|
||||
endpoint: "HTTPS://inference.example.test/api",
|
||||
profileSource: PromptKitConfig{ProfileFile: "./profiles.yml"},
|
||||
},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
cfg := Default()
|
||||
cfg.PromptKit = tt.profileSource
|
||||
cfg.PromptKit.LocalBackend = &PromptKitLocalBackendConfig{
|
||||
Endpoint: tt.endpoint,
|
||||
ConcurrencyLimit: 2,
|
||||
}
|
||||
if err := cfg.Validate(); err != nil {
|
||||
t.Fatalf("Validate() error = %v", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidatePromptKitLocalBackendRejectsInvalidValues(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
endpoint string
|
||||
concurrencyLimit int
|
||||
want string
|
||||
}{
|
||||
{name: "blank endpoint", endpoint: " ", want: "promptkit.local_backend.endpoint"},
|
||||
{name: "relative URL", endpoint: "localhost:8000/v1", want: "promptkit.local_backend.endpoint"},
|
||||
{name: "unsupported scheme", endpoint: "ftp://localhost/model", want: "promptkit.local_backend.endpoint"},
|
||||
{name: "missing host", endpoint: "http:///v1", want: "promptkit.local_backend.endpoint"},
|
||||
{name: "user information", endpoint: "http://user:secret@localhost/v1", want: "promptkit.local_backend.endpoint"},
|
||||
{name: "query", endpoint: "http://localhost/v1?model=example", want: "promptkit.local_backend.endpoint"},
|
||||
{name: "empty query", endpoint: "http://localhost/v1?", want: "promptkit.local_backend.endpoint"},
|
||||
{name: "fragment", endpoint: "http://localhost/v1#model", want: "promptkit.local_backend.endpoint"},
|
||||
{name: "empty fragment", endpoint: "http://localhost/v1#", want: "promptkit.local_backend.endpoint"},
|
||||
{
|
||||
name: "negative concurrency",
|
||||
endpoint: "http://localhost:8000/v1",
|
||||
concurrencyLimit: -1,
|
||||
want: "promptkit.local_backend.concurrency_limit",
|
||||
},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
cfg := Default()
|
||||
cfg.PromptKit.LocalBackend = &PromptKitLocalBackendConfig{
|
||||
Endpoint: tt.endpoint,
|
||||
ConcurrencyLimit: tt.concurrencyLimit,
|
||||
}
|
||||
assertValidationContains(t, cfg, tt.want)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidateStateSurfaceRules(t *testing.T) {
|
||||
|
||||
@@ -2,6 +2,7 @@ package debugbundle
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
@@ -137,6 +138,47 @@ func TestSummaryWriterWritesEverySummaryArtifact(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriteInvocationPreservesReasoningEffortOverrideStates(t *testing.T) {
|
||||
replacement := "focused"
|
||||
cleared := ""
|
||||
tests := []struct {
|
||||
name string
|
||||
override *string
|
||||
wantValue string
|
||||
wantSet bool
|
||||
}{
|
||||
{name: "inherit"},
|
||||
{name: "replace", override: &replacement, wantValue: "focused", wantSet: true},
|
||||
{name: "clear", override: &cleared, wantSet: true},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
bundle, err := Allocate(t.TempDir(), testBundleRunID, time.Unix(0, 42))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := bundle.Summary().WriteInvocation(Invocation{
|
||||
Operation: "run",
|
||||
ReasoningEffortOverride: tt.override,
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
data, err := os.ReadFile(filepath.Join(bundle.SummaryRoot(), ArtifactInvocationMetadata))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var payload map[string]any
|
||||
if err := json.Unmarshal(data, &payload); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
value, found := payload["reasoning_effort_override"]
|
||||
if found != tt.wantSet || (found && value != tt.wantValue) {
|
||||
t.Fatalf("reasoning override found=%t value=%#v, want found=%t value=%q; JSON=%s", found, value, tt.wantSet, tt.wantValue, data)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
func TestSummaryWriterInternalWritesConfineArtifacts(t *testing.T) {
|
||||
bundle, err := Allocate(t.TempDir(), testBundleRunID, time.Unix(0, 42))
|
||||
if err != nil {
|
||||
|
||||
@@ -39,6 +39,7 @@ type Invocation struct {
|
||||
ConfigSource string `json:"config_source,omitempty"`
|
||||
OnlyLanes []string `json:"only_lanes,omitempty"`
|
||||
ChunkCacheOverride string `json:"chunk_cache_override,omitempty"`
|
||||
ReasoningEffortOverride *string `json:"reasoning_effort_override,omitempty"`
|
||||
RunID string `json:"run_id"`
|
||||
StartedAt time.Time `json:"started_at"`
|
||||
}
|
||||
@@ -68,6 +69,10 @@ func (w *SummaryWriter) WriteInvocation(payload Invocation) error {
|
||||
if payload.StartedAt.IsZero() {
|
||||
payload.StartedAt = w.createdAt
|
||||
}
|
||||
if payload.ReasoningEffortOverride != nil {
|
||||
value := *payload.ReasoningEffortOverride
|
||||
payload.ReasoningEffortOverride = &value
|
||||
}
|
||||
return w.writeJSON(ArtifactInvocationMetadata, payload)
|
||||
}
|
||||
func (w *SummaryWriter) WriteRedactedEffectiveConfig(payload RedactedSummaryPayload) error {
|
||||
|
||||
@@ -249,6 +249,9 @@ func TestValidateRefValid(t *testing.T) {
|
||||
if err := ValidateRef(doc, ref); err != nil {
|
||||
t.Fatalf("ValidateRef() error = %v, want nil", err)
|
||||
}
|
||||
if err := NewDocumentIndex(doc).ValidateRef(ref); err != nil {
|
||||
t.Fatalf("DocumentIndex.ValidateRef() error = %v, want nil", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidateRefRejectsMalformedReferences(t *testing.T) {
|
||||
@@ -301,11 +304,60 @@ func TestValidateRefRejectsMalformedReferences(t *testing.T) {
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
err := ValidateRef(validDocument(), tt.ref)
|
||||
|
||||
requireErrorFragments(t, err, tt.fragments...)
|
||||
doc := validDocument()
|
||||
validators := []struct {
|
||||
name string
|
||||
validate func(SourceRef) error
|
||||
}{
|
||||
{name: "document", validate: func(ref SourceRef) error { return ValidateRef(doc, ref) }},
|
||||
{name: "index", validate: NewDocumentIndex(doc).ValidateRef},
|
||||
}
|
||||
for _, validator := range validators {
|
||||
t.Run(validator.name, func(t *testing.T) {
|
||||
requireErrorFragments(t, validator.validate(tt.ref), tt.fragments...)
|
||||
})
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestDocumentIndexSnapshotsIdentityAndUnitPositions(t *testing.T) {
|
||||
doc := &SourceDocument{
|
||||
ID: "source-1",
|
||||
Units: []SourceUnit{
|
||||
{ID: 30},
|
||||
{ID: 10},
|
||||
{ID: 30},
|
||||
},
|
||||
}
|
||||
index := NewDocumentIndex(doc)
|
||||
doc.ID = "changed"
|
||||
doc.Units[0].ID = 99
|
||||
|
||||
if documentID, ok := index.DocumentID(); !ok || documentID != "source-1" {
|
||||
t.Fatalf("DocumentID() = %q, %t, want source-1, true", documentID, ok)
|
||||
}
|
||||
if position, ok := index.Position(30); !ok || position != 0 {
|
||||
t.Fatalf("Position(30) = %d, %t, want 0, true", position, ok)
|
||||
}
|
||||
if position, ok := index.Position(10); !ok || position != 1 {
|
||||
t.Fatalf("Position(10) = %d, %t, want 1, true", position, ok)
|
||||
}
|
||||
ref := SourceRef{SourceID: "source-1", StartUnitID: 30, EndUnitID: 10}
|
||||
if err := index.ValidateRef(ref); err != nil {
|
||||
t.Fatalf("ValidateRef() error = %v, want nil", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestZeroDocumentIndexIsSafe(t *testing.T) {
|
||||
var index DocumentIndex
|
||||
if documentID, ok := index.DocumentID(); ok || documentID != "" {
|
||||
t.Fatalf("DocumentID() = %q, %t, want empty, false", documentID, ok)
|
||||
}
|
||||
if position, ok := index.Position(1); ok || position != 0 {
|
||||
t.Fatalf("Position(1) = %d, %t, want 0, false", position, ok)
|
||||
}
|
||||
requireErrorFragments(t, index.ValidateRef(SourceRef{}), "source document must not be nil")
|
||||
}
|
||||
|
||||
func TestUnitIndex(t *testing.T) {
|
||||
|
||||
@@ -5,6 +5,54 @@ import (
|
||||
"strings"
|
||||
)
|
||||
|
||||
// DocumentIndex is an immutable snapshot of a source document's identity and
|
||||
// unit positions for repeated source-reference operations.
|
||||
type DocumentIndex struct {
|
||||
documentID string
|
||||
positions map[int]int
|
||||
hasDocument bool
|
||||
}
|
||||
|
||||
// NewDocumentIndex snapshots doc without retaining or mutating it.
|
||||
func NewDocumentIndex(doc *SourceDocument) DocumentIndex {
|
||||
if doc == nil {
|
||||
return DocumentIndex{}
|
||||
}
|
||||
positions := make(map[int]int, len(doc.Units))
|
||||
for position, unit := range doc.Units {
|
||||
if _, exists := positions[unit.ID]; !exists {
|
||||
positions[unit.ID] = position
|
||||
}
|
||||
}
|
||||
return DocumentIndex{
|
||||
documentID: doc.ID,
|
||||
positions: positions,
|
||||
hasDocument: true,
|
||||
}
|
||||
}
|
||||
|
||||
// DocumentID returns the indexed document identity.
|
||||
func (i DocumentIndex) DocumentID() (string, bool) {
|
||||
if !i.hasDocument {
|
||||
return "", false
|
||||
}
|
||||
return i.documentID, true
|
||||
}
|
||||
|
||||
// Position returns the indexed document position for unitID.
|
||||
func (i DocumentIndex) Position(unitID int) (int, bool) {
|
||||
position, ok := i.positions[unitID]
|
||||
return position, ok
|
||||
}
|
||||
|
||||
// ValidateRef validates ref against the indexed document snapshot.
|
||||
func (i DocumentIndex) ValidateRef(ref SourceRef) error {
|
||||
if !i.hasDocument {
|
||||
return fmt.Errorf("source document must not be nil")
|
||||
}
|
||||
return validateRef(i.documentID, i.Position, ref)
|
||||
}
|
||||
|
||||
func ValidateDocument(doc *SourceDocument) error {
|
||||
if doc == nil {
|
||||
return fmt.Errorf("source document must not be nil")
|
||||
@@ -60,6 +108,12 @@ func ValidateRef(doc *SourceDocument, ref SourceRef) error {
|
||||
if doc == nil {
|
||||
return fmt.Errorf("source document must not be nil")
|
||||
}
|
||||
return validateRef(doc.ID, func(unitID int) (int, bool) {
|
||||
return UnitIndex(doc, unitID)
|
||||
}, ref)
|
||||
}
|
||||
|
||||
func validateRef(documentID string, position func(int) (int, bool), ref SourceRef) error {
|
||||
if isBlank(ref.SourceID) {
|
||||
return fmt.Errorf("source ref source_id must not be empty")
|
||||
}
|
||||
@@ -72,15 +126,15 @@ func ValidateRef(doc *SourceDocument, ref SourceRef) error {
|
||||
if ref.EndUnitID <= 0 {
|
||||
return fmt.Errorf("source ref end_unit_id must be positive")
|
||||
}
|
||||
if ref.SourceID != doc.ID {
|
||||
return fmt.Errorf("source ref source_id %q does not match document id %q", ref.SourceID, doc.ID)
|
||||
if ref.SourceID != documentID {
|
||||
return fmt.Errorf("source ref source_id %q does not match document id %q", ref.SourceID, documentID)
|
||||
}
|
||||
|
||||
startIndex, ok := UnitIndex(doc, ref.StartUnitID)
|
||||
startIndex, ok := position(ref.StartUnitID)
|
||||
if !ok {
|
||||
return fmt.Errorf("source ref start_unit_id %d was not found", ref.StartUnitID)
|
||||
}
|
||||
endIndex, ok := UnitIndex(doc, ref.EndUnitID)
|
||||
endIndex, ok := position(ref.EndUnitID)
|
||||
if !ok {
|
||||
return fmt.Errorf("source ref end_unit_id %d was not found", ref.EndUnitID)
|
||||
}
|
||||
|
||||
@@ -52,13 +52,13 @@ func (l *FilesystemLoader) Source(moduleKey string) (pipeline.SourceCheckpoint,
|
||||
}
|
||||
doc := cloneSourceDocument(payload.Document)
|
||||
if err := source.ValidateDocument(&doc); err != nil {
|
||||
return pipeline.SourceCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactPayloadInvalid, "source checkpoint document is invalid")
|
||||
return pipeline.SourceCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactPayloadInvalid)
|
||||
}
|
||||
if strings.TrimSpace(manifest.SourceID) != "" && manifest.SourceID != doc.ID {
|
||||
return pipeline.SourceCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactPayloadInvalid, "source checkpoint identity does not match its payload")
|
||||
return pipeline.SourceCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactPayloadInvalid)
|
||||
}
|
||||
if !fingerprintsEqual(checkpointToPipelineFingerprints(manifest.OutputDigests), digestFingerprints("source_document", doc.Digest)) {
|
||||
return pipeline.SourceCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactDigestMismatch, "source checkpoint output digest does not match its payload")
|
||||
return pipeline.SourceCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactDigestMismatch)
|
||||
}
|
||||
return pipeline.SourceCheckpoint{Document: &doc}, reusedDecision()
|
||||
}
|
||||
@@ -84,7 +84,7 @@ func (l *FilesystemLoader) ExtractForStep(stepID, laneID, moduleKey string, depe
|
||||
return pipeline.ExtractCheckpoint{}, artifactDecision(err, "extract checkpoint artifact is invalid")
|
||||
}
|
||||
if !fingerprintsEqual(checkpointToPipelineFingerprints(manifest.OutputDigests), artifactOutputDigests(outputs)) {
|
||||
return pipeline.ExtractCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactDigestMismatch, "extract checkpoint output digest does not match its payload")
|
||||
return pipeline.ExtractCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactDigestMismatch)
|
||||
}
|
||||
return pipeline.ExtractCheckpoint{Outputs: outputs, Rejected: cloneRejectedOutputs(payload.Rejected), Warnings: cloneWarnings(payload.Warnings)}, reusedDecision()
|
||||
}
|
||||
@@ -110,7 +110,7 @@ func (l *FilesystemLoader) MergeForStep(stepID, laneID, moduleKey string, depend
|
||||
return pipeline.MergeCheckpoint{}, artifactDecision(err, "merge checkpoint artifact is invalid")
|
||||
}
|
||||
if !fingerprintsEqual(checkpointToPipelineFingerprints(manifest.OutputDigests), artifactOutputDigests(values)) {
|
||||
return pipeline.MergeCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactDigestMismatch, "merge checkpoint output digest does not match its payload")
|
||||
return pipeline.MergeCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactDigestMismatch)
|
||||
}
|
||||
return pipeline.MergeCheckpoint{Output: values[0], Warnings: cloneWarnings(payload.Warnings)}, reusedDecision()
|
||||
}
|
||||
@@ -136,7 +136,7 @@ func (l *FilesystemLoader) NormalizeForStep(stepID, laneID, moduleKey string, de
|
||||
return pipeline.NormalizeCheckpoint{}, artifactDecision(err, "normalize checkpoint artifact is invalid")
|
||||
}
|
||||
if !fingerprintsEqual(checkpointToPipelineFingerprints(manifest.OutputDigests), artifactOutputDigests(values)) {
|
||||
return pipeline.NormalizeCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactDigestMismatch, "normalize checkpoint output digest does not match its payload")
|
||||
return pipeline.NormalizeCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactDigestMismatch)
|
||||
}
|
||||
return pipeline.NormalizeCheckpoint{Output: values[0], Warnings: cloneWarnings(payload.Warnings)}, reusedDecision()
|
||||
}
|
||||
@@ -158,36 +158,36 @@ func (l *FilesystemLoader) AcceptedNormalize(stepID, laneID, moduleKey string) (
|
||||
return pipeline.NormalizeCheckpoint{}, artifactDecision(err, "accepted normalize checkpoint artifact is invalid")
|
||||
}
|
||||
if len(values) != 1 {
|
||||
return pipeline.NormalizeCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactPayloadInvalid, "accepted normalize checkpoint payload is invalid")
|
||||
return pipeline.NormalizeCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactPayloadInvalid)
|
||||
}
|
||||
if !fingerprintsEqual(checkpointToPipelineFingerprints(manifest.OutputDigests), artifactOutputDigests(values)) {
|
||||
return pipeline.NormalizeCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactDigestMismatch, "accepted normalize checkpoint digest does not match its payload")
|
||||
return pipeline.NormalizeCheckpoint{}, decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonArtifactDigestMismatch)
|
||||
}
|
||||
return pipeline.NormalizeCheckpoint{Output: values[0], Warnings: cloneWarnings(payload.Warnings)}, decision(pipeline.CheckpointDecisionReused, pipeline.CheckpointReasonAcceptedArtifactReused, "accepted normalized artifact is reusable")
|
||||
return pipeline.NormalizeCheckpoint{Output: values[0], Warnings: cloneWarnings(payload.Warnings)}, decision(pipeline.CheckpointDecisionReused, pipeline.CheckpointReasonAcceptedArtifactReused)
|
||||
}
|
||||
|
||||
func (l *FilesystemLoader) validateAcceptedNormalizeManifest(manifest StageManifest, stepID, laneID, moduleKey string) pipeline.CheckpointDecision {
|
||||
if manifest.WorkspaceSchemaVersion != WorkspaceSchemaVersion {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonWorkspaceSchemaIncompatible, "checkpoint workspace schema is incompatible")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonWorkspaceSchemaIncompatible)
|
||||
}
|
||||
identity := strings.TrimSpace(l.identityDigest)
|
||||
if identity == "" || strings.TrimSpace(manifest.Metadata["checkpoint_identity_digest"]) == "" || manifest.Metadata["checkpoint_identity_digest"] != identity {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonIdentityMismatch, "checkpoint identity is unavailable or does not match the current invocation")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonIdentityMismatch)
|
||||
}
|
||||
if manifest.Stage != StageNormalize {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonStageMismatch, "checkpoint stage does not match normalize")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonStageMismatch)
|
||||
}
|
||||
if strings.TrimSpace(stepID) == "" || manifest.StepID != stepID {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonStepMismatch, "checkpoint step does not match the requested step")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonStepMismatch)
|
||||
}
|
||||
if strings.TrimSpace(laneID) == "" || manifest.LaneID != laneID {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonLaneMismatch, "checkpoint lane does not match the requested lane")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonLaneMismatch)
|
||||
}
|
||||
if strings.TrimSpace(moduleKey) == "" || manifest.ModuleKey != moduleKey {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonModuleMismatch, "checkpoint module does not match the requested normalizer")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonModuleMismatch)
|
||||
}
|
||||
if manifest.Status != StatusSucceeded {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonStatusNotReusable, "checkpoint status cannot provide an accepted normalized artifact")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonStatusNotReusable)
|
||||
}
|
||||
return reusedDecision()
|
||||
}
|
||||
@@ -212,21 +212,21 @@ func artifactCheckpointOutputs(values []artifactCheckpointEnvelope) ([]pipeline.
|
||||
|
||||
func (l *FilesystemLoader) readJSON(name string, out any) pipeline.CheckpointDecision {
|
||||
if !l.Enabled() {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonLoadingDisabled, "checkpoint loading disabled")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonLoadingDisabled)
|
||||
}
|
||||
target, err := fileio.SafePath(l.root, name)
|
||||
if err != nil {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonPathInvalid, "checkpoint path is invalid")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonPathInvalid)
|
||||
}
|
||||
data, err := os.ReadFile(target)
|
||||
if err != nil {
|
||||
if os.IsNotExist(err) {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonMissing, "checkpoint artifact is missing")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonMissing)
|
||||
}
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonReadFailed, "checkpoint artifact could not be read")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonReadFailed)
|
||||
}
|
||||
if err := json.Unmarshal(data, out); err != nil {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonDecodeFailed, "checkpoint artifact could not be decoded")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonDecodeFailed)
|
||||
}
|
||||
return reusedDecision()
|
||||
}
|
||||
@@ -237,28 +237,28 @@ func (l *FilesystemLoader) validateManifest(manifest StageManifest, stage StageN
|
||||
|
||||
func (l *FilesystemLoader) validateLaneManifest(manifest StageManifest, stage StageName, stepID string, laneID string, moduleKey string, dependencies []pipeline.CheckpointFingerprint, statuses ...StageStatus) pipeline.CheckpointDecision {
|
||||
if manifest.WorkspaceSchemaVersion == WorkspaceSchemaVersionV1 {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonWorkspaceSchemaIncompatible, "checkpoint workspace schema is incompatible")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonWorkspaceSchemaIncompatible)
|
||||
}
|
||||
if manifest.WorkspaceSchemaVersion == WorkspaceSchemaVersionV2 {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonWorkspaceSchemaIncompatible, "checkpoint workspace schema is incompatible")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonWorkspaceSchemaIncompatible)
|
||||
}
|
||||
if manifest.WorkspaceSchemaVersion != WorkspaceSchemaVersion {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonWorkspaceSchemaIncompatible, "checkpoint workspace schema is incompatible")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonWorkspaceSchemaIncompatible)
|
||||
}
|
||||
if strings.TrimSpace(l.identityDigest) != "" && manifest.Metadata["checkpoint_identity_digest"] != l.identityDigest {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonIdentityMismatch, "checkpoint identity does not match the current invocation")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonIdentityMismatch)
|
||||
}
|
||||
if manifest.Stage != stage {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonStageMismatch, "checkpoint stage does not match the requested stage")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonStageMismatch)
|
||||
}
|
||||
if strings.TrimSpace(stepID) != "" && manifest.StepID != stepID {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonStepMismatch, "checkpoint step does not match the requested step")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonStepMismatch)
|
||||
}
|
||||
if strings.TrimSpace(laneID) != "" && manifest.LaneID != laneID {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonLaneMismatch, "checkpoint lane does not match the requested lane")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonLaneMismatch)
|
||||
}
|
||||
if strings.TrimSpace(moduleKey) != "" && manifest.ModuleKey != moduleKey {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonModuleMismatch, "checkpoint module does not match the requested module")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonModuleMismatch)
|
||||
}
|
||||
statusOK := false
|
||||
for _, status := range statuses {
|
||||
@@ -268,10 +268,10 @@ func (l *FilesystemLoader) validateLaneManifest(manifest StageManifest, stage St
|
||||
}
|
||||
}
|
||||
if !statusOK {
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonStatusNotReusable, "checkpoint status cannot be reused")
|
||||
return decision(pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonStatusNotReusable)
|
||||
}
|
||||
if !fingerprintsEqual(checkpointToPipelineFingerprints(manifest.DependencyFingerprints), dependencies) {
|
||||
return decision(pipeline.CheckpointDecisionDependencyInvalidated, pipeline.CheckpointReasonDependencyMismatch, "checkpoint dependencies do not match")
|
||||
return decision(pipeline.CheckpointDecisionDependencyInvalidated, pipeline.CheckpointReasonDependencyMismatch)
|
||||
}
|
||||
return reusedDecision()
|
||||
}
|
||||
@@ -313,11 +313,11 @@ func fingerprintsEqual(a []pipeline.CheckpointFingerprint, b []pipeline.Checkpoi
|
||||
}
|
||||
|
||||
func reusedDecision() pipeline.CheckpointDecision {
|
||||
return decision(pipeline.CheckpointDecisionReused, pipeline.CheckpointReasonReused, "checkpoint is reusable")
|
||||
return decision(pipeline.CheckpointDecisionReused, pipeline.CheckpointReasonReused)
|
||||
}
|
||||
|
||||
func decision(category pipeline.CheckpointDecisionCategory, code pipeline.CheckpointReasonCode, detail string) pipeline.CheckpointDecision {
|
||||
return pipeline.NewCheckpointDecision(category, code, detail)
|
||||
func decision(category pipeline.CheckpointDecisionCategory, code pipeline.CheckpointReasonCode) pipeline.CheckpointDecision {
|
||||
return pipeline.NewCheckpointDecision(category, code)
|
||||
}
|
||||
|
||||
type artifactPayloadError struct {
|
||||
@@ -332,5 +332,5 @@ func artifactDecision(err error, detail string) pipeline.CheckpointDecision {
|
||||
if payloadErr, ok := err.(*artifactPayloadError); ok {
|
||||
code = payloadErr.code
|
||||
}
|
||||
return decision(pipeline.CheckpointDecisionExecuted, code, detail)
|
||||
return decision(pipeline.CheckpointDecisionExecuted, code)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
{
|
||||
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
||||
"$id": "notarius.source.chunk_map",
|
||||
"title": "notarius_source_chunk_map_v1",
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": [
|
||||
"source_id",
|
||||
"source_digest",
|
||||
"plan_digest",
|
||||
"requested_chunker",
|
||||
"producer",
|
||||
"plan_annotations",
|
||||
"chunks"
|
||||
],
|
||||
"properties": {
|
||||
"source_id": {"type": "string", "minLength": 1},
|
||||
"source_digest": {"type": "string", "pattern": "^sha256:[0-9a-f]{64}$"},
|
||||
"plan_digest": {"type": "string", "pattern": "^sha256:[0-9a-f]{64}$"},
|
||||
"requested_chunker": {"type": "string", "minLength": 1},
|
||||
"producer": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["input_module", "chunk_module"],
|
||||
"properties": {
|
||||
"input_module": {"type": "string", "minLength": 1},
|
||||
"chunk_module": {"type": "string", "minLength": 1},
|
||||
"llm_profile": {"type": "string", "minLength": 1}
|
||||
}
|
||||
},
|
||||
"plan_annotations": {"$ref": "#/$defs/annotations"},
|
||||
"chunks": {
|
||||
"type": "array",
|
||||
"minItems": 1,
|
||||
"items": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["id", "index", "source_ref", "unit_count", "annotations"],
|
||||
"properties": {
|
||||
"id": {"type": "string", "minLength": 1},
|
||||
"index": {"type": "integer", "minimum": 0},
|
||||
"source_ref": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["source_id", "start_unit_id", "end_unit_id"],
|
||||
"properties": {
|
||||
"source_id": {"type": "string", "minLength": 1},
|
||||
"start_unit_id": {"type": "integer", "minimum": 1},
|
||||
"end_unit_id": {"type": "integer", "minimum": 1}
|
||||
}
|
||||
},
|
||||
"unit_count": {"type": "integer", "minimum": 1},
|
||||
"annotations": {"$ref": "#/$defs/annotations"}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"$defs": {
|
||||
"annotations": {
|
||||
"type": "object",
|
||||
"propertyNames": {"type": "string", "minLength": 1},
|
||||
"additionalProperties": true
|
||||
}
|
||||
}
|
||||
}
|
||||
400
internal/framework/chunkmap/codec.go
Normal file
400
internal/framework/chunkmap/codec.go
Normal file
@@ -0,0 +1,400 @@
|
||||
package chunkmap
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"embed"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"regexp"
|
||||
"strings"
|
||||
"sync"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"github.com/santhosh-tekuri/jsonschema/v6"
|
||||
)
|
||||
|
||||
//go:embed assets/schemas/source_chunk_map.v1.json
|
||||
var schemaAssets embed.FS
|
||||
|
||||
var digestPattern = regexp.MustCompile(`^sha256:[0-9a-f]{64}$`)
|
||||
|
||||
var (
|
||||
loadSchemaOnce sync.Once
|
||||
loadedSchema []byte
|
||||
compiledSchema *jsonschema.Schema
|
||||
loadSchemaErr error
|
||||
)
|
||||
|
||||
// Codec owns strict serialization for the durable chunk-map contract.
|
||||
type Codec struct{}
|
||||
|
||||
func New() *Codec { return &Codec{} }
|
||||
|
||||
func (c *Codec) Kind() contracts.ArtifactKind { return ArtifactKind }
|
||||
|
||||
func (c *Codec) Schema() contracts.ArtifactSchema {
|
||||
raw, err := c.schemaBytes()
|
||||
if err != nil {
|
||||
return contracts.ArtifactSchema{}
|
||||
}
|
||||
return contracts.ArtifactSchema{
|
||||
ID: SchemaID,
|
||||
Name: SchemaName,
|
||||
Version: SchemaVersion,
|
||||
JSONSchema: raw,
|
||||
}
|
||||
}
|
||||
|
||||
func (c *Codec) MediaType() string { return MediaType }
|
||||
|
||||
// Build proves that a durable value describes the exact accepted source plan
|
||||
// and materialized chunk list supplied by the framework.
|
||||
func Build(request BuildRequest) (ChunkMap, error) {
|
||||
if err := source.ValidateDocument(request.Source); err != nil {
|
||||
return ChunkMap{}, fmt.Errorf("validate source document: %w", err)
|
||||
}
|
||||
sourceDigest, err := source.DigestDocument(request.Source)
|
||||
if err != nil {
|
||||
return ChunkMap{}, fmt.Errorf("digest source document: %w", err)
|
||||
}
|
||||
if sourceDigest != request.Source.Digest {
|
||||
return ChunkMap{}, fmt.Errorf("source digest %q does not match source document digest %q", sourceDigest, request.Source.Digest)
|
||||
}
|
||||
if sourceDigest != request.Plan.SourceDigest {
|
||||
return ChunkMap{}, fmt.Errorf("source digest %q does not match chunk plan source digest %q", sourceDigest, request.Plan.SourceDigest)
|
||||
}
|
||||
plan, err := source.CanonicalizeChunkPlan(request.Plan)
|
||||
if err != nil {
|
||||
return ChunkMap{}, fmt.Errorf("canonicalize chunk plan: %w", err)
|
||||
}
|
||||
if err := source.ValidateChunkPlan(request.Source, plan); err != nil {
|
||||
return ChunkMap{}, fmt.Errorf("validate accepted chunk plan: %w", err)
|
||||
}
|
||||
planDigest, err := source.DigestChunkPlan(plan)
|
||||
if err != nil {
|
||||
return ChunkMap{}, fmt.Errorf("digest accepted chunk plan: %w", err)
|
||||
}
|
||||
expected, err := source.MaterializeChunkPlan(request.Source, plan)
|
||||
if err != nil {
|
||||
return ChunkMap{}, fmt.Errorf("materialize accepted chunk plan: %w", err)
|
||||
}
|
||||
if err := verifyMaterializedChunks(request.Chunks, expected); err != nil {
|
||||
return ChunkMap{}, err
|
||||
}
|
||||
|
||||
value := ChunkMap{
|
||||
SourceID: request.Source.ID,
|
||||
SourceDigest: sourceDigest,
|
||||
PlanDigest: planDigest,
|
||||
RequestedChunker: request.RequestedChunker,
|
||||
Producer: request.Producer,
|
||||
PlanAnnotations: source.CloneChunkAnnotations(plan.Annotations),
|
||||
Chunks: make([]Chunk, len(expected)),
|
||||
}
|
||||
for index, chunk := range expected {
|
||||
value.Chunks[index] = Chunk{
|
||||
ID: chunk.ID,
|
||||
Index: chunk.Index,
|
||||
SourceRef: chunk.Ref,
|
||||
UnitCount: len(chunk.Units),
|
||||
Annotations: source.CloneChunkAnnotations(chunk.Annotations),
|
||||
}
|
||||
}
|
||||
canonical, err := canonicalize(value)
|
||||
if err != nil {
|
||||
return ChunkMap{}, fmt.Errorf("validate chunk map: %w", err)
|
||||
}
|
||||
return clone(canonical), nil
|
||||
}
|
||||
|
||||
// Serialize builds and encodes the framework-owned serialized artifact.
|
||||
func Serialize(request BuildRequest) (contracts.SerializedArtifact, error) {
|
||||
value, err := Build(request)
|
||||
if err != nil {
|
||||
return contracts.SerializedArtifact{}, err
|
||||
}
|
||||
codec := New()
|
||||
content, err := codec.Encode(value)
|
||||
if err != nil {
|
||||
return contracts.SerializedArtifact{}, err
|
||||
}
|
||||
return contracts.SerializedArtifact{
|
||||
Kind: ArtifactKind,
|
||||
Schema: codec.Schema(),
|
||||
MediaType: MediaType,
|
||||
Content: content,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func (c *Codec) Encode(value ChunkMap) ([]byte, error) {
|
||||
if _, err := c.schemaBytes(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
canonical, err := canonicalize(clone(value))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("encode source chunk map: %w", err)
|
||||
}
|
||||
content, err := json.Marshal(canonical)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("encode source chunk map: %w", err)
|
||||
}
|
||||
if err := validateSchemaInstance(content); err != nil {
|
||||
return nil, fmt.Errorf("encode source chunk map: %w", err)
|
||||
}
|
||||
return content, nil
|
||||
}
|
||||
|
||||
func (c *Codec) Decode(content []byte) (ChunkMap, error) {
|
||||
if _, err := c.schemaBytes(); err != nil {
|
||||
return ChunkMap{}, err
|
||||
}
|
||||
if err := validateSchemaInstance(content); err != nil {
|
||||
return ChunkMap{}, fmt.Errorf("decode source chunk map: %w", err)
|
||||
}
|
||||
decoder := json.NewDecoder(bytes.NewReader(content))
|
||||
decoder.DisallowUnknownFields()
|
||||
var value ChunkMap
|
||||
if err := decoder.Decode(&value); err != nil {
|
||||
return ChunkMap{}, fmt.Errorf("decode source chunk map: %w", err)
|
||||
}
|
||||
var trailing any
|
||||
if err := decoder.Decode(&trailing); err != io.EOF {
|
||||
return ChunkMap{}, fmt.Errorf("decode source chunk map: multiple JSON values")
|
||||
}
|
||||
canonical, err := canonicalize(value)
|
||||
if err != nil {
|
||||
return ChunkMap{}, fmt.Errorf("decode source chunk map: %w", err)
|
||||
}
|
||||
return clone(canonical), nil
|
||||
}
|
||||
|
||||
func (c *Codec) schemaBytes() ([]byte, error) {
|
||||
loadSchemaOnce.Do(loadAndCompileSchema)
|
||||
if loadSchemaErr != nil {
|
||||
return nil, loadSchemaErr
|
||||
}
|
||||
return append([]byte(nil), loadedSchema...), nil
|
||||
}
|
||||
|
||||
func loadAndCompileSchema() {
|
||||
raw, err := schemaAssets.ReadFile("assets/schemas/source_chunk_map.v1.json")
|
||||
if err != nil {
|
||||
loadSchemaErr = fmt.Errorf("read source chunk map schema: %w", err)
|
||||
return
|
||||
}
|
||||
var identity struct {
|
||||
ID string `json:"$id"`
|
||||
Title string `json:"title"`
|
||||
Type string `json:"type"`
|
||||
Required []string `json:"required"`
|
||||
}
|
||||
if err := json.Unmarshal(raw, &identity); err != nil {
|
||||
loadSchemaErr = fmt.Errorf("decode source chunk map schema: %w", err)
|
||||
return
|
||||
}
|
||||
if identity.ID != SchemaID || identity.Title != SchemaName || identity.Type != "object" || !hasRequiredFields(identity.Required) {
|
||||
loadSchemaErr = fmt.Errorf("source chunk map schema identity or required fields are invalid")
|
||||
return
|
||||
}
|
||||
schemaDocument, err := jsonschema.UnmarshalJSON(bytes.NewReader(raw))
|
||||
if err != nil {
|
||||
loadSchemaErr = fmt.Errorf("parse source chunk map schema: %w", err)
|
||||
return
|
||||
}
|
||||
compiler := jsonschema.NewCompiler()
|
||||
if err := compiler.AddResource("source-chunk-map-schema.json", schemaDocument); err != nil {
|
||||
loadSchemaErr = fmt.Errorf("load source chunk map schema: %w", err)
|
||||
return
|
||||
}
|
||||
compiled, err := compiler.Compile("source-chunk-map-schema.json")
|
||||
if err != nil {
|
||||
loadSchemaErr = fmt.Errorf("compile source chunk map schema: %w", err)
|
||||
return
|
||||
}
|
||||
loadedSchema = append([]byte(nil), raw...)
|
||||
compiledSchema = compiled
|
||||
}
|
||||
|
||||
func validateSchemaInstance(content []byte) error {
|
||||
instance, err := jsonschema.UnmarshalJSON(bytes.NewReader(content))
|
||||
if err != nil {
|
||||
return fmt.Errorf("payload is not valid JSON: %w", err)
|
||||
}
|
||||
if err := compiledSchema.Validate(instance); err != nil {
|
||||
return fmt.Errorf("payload does not conform to source chunk map schema: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func hasRequiredFields(required []string) bool {
|
||||
want := map[string]bool{
|
||||
"source_id": true, "source_digest": true, "plan_digest": true,
|
||||
"requested_chunker": true, "producer": true, "plan_annotations": true,
|
||||
"chunks": true,
|
||||
}
|
||||
for _, field := range required {
|
||||
delete(want, field)
|
||||
}
|
||||
return len(want) == 0
|
||||
}
|
||||
|
||||
func canonicalize(value ChunkMap) (ChunkMap, error) {
|
||||
if err := requireIdentity("source_id", value.SourceID); err != nil {
|
||||
return ChunkMap{}, err
|
||||
}
|
||||
if err := requireDigest("source_digest", value.SourceDigest); err != nil {
|
||||
return ChunkMap{}, err
|
||||
}
|
||||
if err := requireDigest("plan_digest", value.PlanDigest); err != nil {
|
||||
return ChunkMap{}, err
|
||||
}
|
||||
if err := requireIdentity("requested_chunker", value.RequestedChunker); err != nil {
|
||||
return ChunkMap{}, err
|
||||
}
|
||||
if err := requireIdentity("producer.input_module", value.Producer.InputModule); err != nil {
|
||||
return ChunkMap{}, err
|
||||
}
|
||||
if err := requireIdentity("producer.chunk_module", value.Producer.ChunkModule); err != nil {
|
||||
return ChunkMap{}, err
|
||||
}
|
||||
if value.Producer.LLMProfile != "" {
|
||||
if err := requireIdentity("producer.llm_profile", value.Producer.LLMProfile); err != nil {
|
||||
return ChunkMap{}, err
|
||||
}
|
||||
}
|
||||
annotations, err := canonicalizeAnnotations("plan_annotations", value.PlanAnnotations)
|
||||
if err != nil {
|
||||
return ChunkMap{}, err
|
||||
}
|
||||
value.PlanAnnotations = annotations
|
||||
if len(value.Chunks) == 0 {
|
||||
return ChunkMap{}, fmt.Errorf("chunks must not be empty")
|
||||
}
|
||||
seenIDs := make(map[string]struct{}, len(value.Chunks))
|
||||
plan := source.ChunkPlan{SourceDigest: value.SourceDigest, Annotations: annotations, Ranges: make([]source.ChunkRange, len(value.Chunks))}
|
||||
for index := range value.Chunks {
|
||||
chunk := &value.Chunks[index]
|
||||
if err := requireIdentity(fmt.Sprintf("chunks[%d].id", index), chunk.ID); err != nil {
|
||||
return ChunkMap{}, err
|
||||
}
|
||||
if _, exists := seenIDs[chunk.ID]; exists {
|
||||
return ChunkMap{}, fmt.Errorf("chunks[%d].id %q is duplicated", index, chunk.ID)
|
||||
}
|
||||
seenIDs[chunk.ID] = struct{}{}
|
||||
if chunk.Index != index {
|
||||
return ChunkMap{}, fmt.Errorf("chunks[%d].index = %d, want %d", index, chunk.Index, index)
|
||||
}
|
||||
if chunk.SourceRef.SourceID != value.SourceID {
|
||||
return ChunkMap{}, fmt.Errorf("chunks[%d].source_ref.source_id %q does not match source_id %q", index, chunk.SourceRef.SourceID, value.SourceID)
|
||||
}
|
||||
if chunk.SourceRef.StartUnitID <= 0 || chunk.SourceRef.EndUnitID <= 0 {
|
||||
return ChunkMap{}, fmt.Errorf("chunks[%d].source_ref endpoints must be positive", index)
|
||||
}
|
||||
if chunk.UnitCount <= 0 {
|
||||
return ChunkMap{}, fmt.Errorf("chunks[%d].unit_count must be positive", index)
|
||||
}
|
||||
chunkAnnotations, err := canonicalizeAnnotations(fmt.Sprintf("chunks[%d].annotations", index), chunk.Annotations)
|
||||
if err != nil {
|
||||
return ChunkMap{}, err
|
||||
}
|
||||
chunk.Annotations = chunkAnnotations
|
||||
plan.Ranges[index] = source.ChunkRange{
|
||||
StartUnitID: chunk.SourceRef.StartUnitID,
|
||||
EndUnitID: chunk.SourceRef.EndUnitID,
|
||||
Annotations: chunkAnnotations,
|
||||
}
|
||||
}
|
||||
planDigest, err := source.DigestChunkPlan(plan)
|
||||
if err != nil {
|
||||
return ChunkMap{}, fmt.Errorf("reconstruct plan digest: %w", err)
|
||||
}
|
||||
if planDigest != value.PlanDigest {
|
||||
return ChunkMap{}, fmt.Errorf("plan_digest %q does not match reconstructed plan digest %q", value.PlanDigest, planDigest)
|
||||
}
|
||||
return value, nil
|
||||
}
|
||||
|
||||
func canonicalizeAnnotations(name string, annotations source.ChunkAnnotations) (source.ChunkAnnotations, error) {
|
||||
for namespace := range annotations {
|
||||
if strings.TrimSpace(namespace) == "" || namespace != strings.TrimSpace(namespace) {
|
||||
return nil, fmt.Errorf("%s namespace %q must be non-empty and trimmed", name, namespace)
|
||||
}
|
||||
}
|
||||
canonical, err := source.CanonicalizeChunkAnnotations(annotations)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", name, err)
|
||||
}
|
||||
if canonical == nil {
|
||||
canonical = source.ChunkAnnotations{}
|
||||
}
|
||||
return canonical, nil
|
||||
}
|
||||
|
||||
func requireIdentity(name, value string) error {
|
||||
if strings.TrimSpace(value) == "" || value != strings.TrimSpace(value) {
|
||||
return fmt.Errorf("%s must be non-empty and trimmed", name)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func requireDigest(name, value string) error {
|
||||
if !digestPattern.MatchString(value) {
|
||||
return fmt.Errorf("%s must be a canonical sha256 digest", name)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func verifyMaterializedChunks(actual, expected []source.Chunk) error {
|
||||
if len(actual) != len(expected) {
|
||||
return fmt.Errorf("materialized chunks length = %d, want %d", len(actual), len(expected))
|
||||
}
|
||||
for index := range expected {
|
||||
got, want := actual[index], expected[index]
|
||||
if got.ID != want.ID || got.SourceID != want.SourceID || got.Index != want.Index || got.Ref != want.Ref {
|
||||
return fmt.Errorf("materialized chunk[%d] identity or source range differs from accepted plan", index)
|
||||
}
|
||||
if len(got.Units) != len(want.Units) || !sameUnits(got.Units, want.Units) {
|
||||
return fmt.Errorf("materialized chunk[%d] units differ from accepted source range", index)
|
||||
}
|
||||
if !sameAnnotations(got.PlanAnnotations, want.PlanAnnotations) || !sameAnnotations(got.Annotations, want.Annotations) {
|
||||
return fmt.Errorf("materialized chunk[%d] annotations differ from accepted plan", index)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func sameUnits(left, right []source.SourceUnit) bool {
|
||||
leftJSON, leftErr := json.Marshal(left)
|
||||
rightJSON, rightErr := json.Marshal(right)
|
||||
return leftErr == nil && rightErr == nil && bytes.Equal(leftJSON, rightJSON)
|
||||
}
|
||||
|
||||
func sameAnnotations(left, right source.ChunkAnnotations) bool {
|
||||
leftCanonical, leftErr := source.CanonicalizeChunkAnnotations(left)
|
||||
rightCanonical, rightErr := source.CanonicalizeChunkAnnotations(right)
|
||||
if leftErr != nil || rightErr != nil {
|
||||
return false
|
||||
}
|
||||
leftJSON, leftErr := json.Marshal(leftCanonical)
|
||||
rightJSON, rightErr := json.Marshal(rightCanonical)
|
||||
return leftErr == nil && rightErr == nil && bytes.Equal(leftJSON, rightJSON)
|
||||
}
|
||||
|
||||
func clone(value ChunkMap) ChunkMap {
|
||||
value.PlanAnnotations = cloneAnnotations(value.PlanAnnotations)
|
||||
value.Chunks = append([]Chunk(nil), value.Chunks...)
|
||||
for index := range value.Chunks {
|
||||
value.Chunks[index].Annotations = cloneAnnotations(value.Chunks[index].Annotations)
|
||||
}
|
||||
return value
|
||||
}
|
||||
|
||||
func cloneAnnotations(annotations source.ChunkAnnotations) source.ChunkAnnotations {
|
||||
cloned := source.CloneChunkAnnotations(annotations)
|
||||
if cloned == nil {
|
||||
return source.ChunkAnnotations{}
|
||||
}
|
||||
return cloned
|
||||
}
|
||||
263
internal/framework/chunkmap/codec_test.go
Normal file
263
internal/framework/chunkmap/codec_test.go
Normal file
@@ -0,0 +1,263 @@
|
||||
package chunkmap
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"os"
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
)
|
||||
|
||||
func TestBuildAndSerializeAcceptedChunkMap(t *testing.T) {
|
||||
request := acceptedBuildRequest(t)
|
||||
value, err := Build(request)
|
||||
if err != nil {
|
||||
t.Fatalf("Build() error = %v", err)
|
||||
}
|
||||
if value.SourceID != request.Source.ID || len(value.Chunks) != 2 || value.Chunks[0].UnitCount != 2 || value.Chunks[1].SourceRef.StartUnitID != 20 {
|
||||
t.Fatalf("Build() = %#v, want exact accepted chunk structure", value)
|
||||
}
|
||||
if value.PlanAnnotations == nil || value.Chunks[1].Annotations == nil {
|
||||
t.Fatalf("Build() annotations = %#v, want explicit maps", value)
|
||||
}
|
||||
artifact, err := Serialize(request)
|
||||
if err != nil {
|
||||
t.Fatalf("Serialize() error = %v", err)
|
||||
}
|
||||
if artifact.Kind != ArtifactKind || artifact.Schema.ID != SchemaID || artifact.Schema.Name != SchemaName || artifact.Schema.Version != SchemaVersion || artifact.MediaType != MediaType || artifact.Metadata != nil {
|
||||
t.Fatalf("Serialize() = %#v, want fixed artifact envelope without metadata", artifact)
|
||||
}
|
||||
decoded, err := New().Decode(artifact.Content)
|
||||
if err != nil {
|
||||
t.Fatalf("Decode(Serialize()) error = %v", err)
|
||||
}
|
||||
if decoded.PlanDigest != value.PlanDigest || decoded.Chunks[0].ID != "chunk-000001" || decoded.Chunks[1].UnitCount != 1 {
|
||||
t.Fatalf("Decode(Serialize()) = %#v, want durable chunk map", decoded)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCodecRoundTripsValidFixture(t *testing.T) {
|
||||
fixture, err := os.ReadFile("testdata/source_chunk_map.v1.json")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
codec := New()
|
||||
value, err := codec.Decode(fixture)
|
||||
if err != nil {
|
||||
t.Fatalf("Decode(fixture) error = %v", err)
|
||||
}
|
||||
encoded, err := codec.Encode(value)
|
||||
if err != nil {
|
||||
t.Fatalf("Encode(decoded fixture) error = %v", err)
|
||||
}
|
||||
if !bytes.Equal(encoded, bytes.TrimSpace(fixture)) {
|
||||
t.Fatalf("fixture does not use canonical encoding\nwant: %s\n got: %s", fixture, encoded)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildCanonicalizesAnnotationFormatting(t *testing.T) {
|
||||
first := acceptedBuildRequest(t)
|
||||
second := acceptedBuildRequest(t)
|
||||
second.Plan.Annotations["test/chunker"] = json.RawMessage(" { \n \t\"label\" : \"fixture\" \n } ")
|
||||
canonical, err := source.CanonicalizeChunkPlan(second.Plan)
|
||||
if err != nil {
|
||||
t.Fatalf("CanonicalizeChunkPlan() error = %v", err)
|
||||
}
|
||||
second.Chunks, err = source.MaterializeChunkPlan(second.Source, canonical)
|
||||
if err != nil {
|
||||
t.Fatalf("MaterializeChunkPlan() error = %v", err)
|
||||
}
|
||||
firstArtifact, err := Serialize(first)
|
||||
if err != nil {
|
||||
t.Fatalf("Serialize(first) error = %v", err)
|
||||
}
|
||||
secondArtifact, err := Serialize(second)
|
||||
if err != nil {
|
||||
t.Fatalf("Serialize(second) error = %v", err)
|
||||
}
|
||||
if !bytes.Equal(firstArtifact.Content, secondArtifact.Content) {
|
||||
t.Fatalf("serialized content differs only because annotation whitespace changed\nfirst: %s\nsecond: %s", firstArtifact.Content, secondArtifact.Content)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildRejectsChunksOutsideAcceptedPlan(t *testing.T) {
|
||||
request := acceptedBuildRequest(t)
|
||||
request.Chunks[0].Units[0].ID = 999
|
||||
if _, err := Build(request); err == nil {
|
||||
t.Fatal("Build() error = nil, want rejection for chunk units outside accepted source range")
|
||||
}
|
||||
}
|
||||
|
||||
func TestCodecRejectsInvalidDurableBoundaries(t *testing.T) {
|
||||
value, err := Build(acceptedBuildRequest(t))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
mutate func(*ChunkMap)
|
||||
}{
|
||||
{name: "blank identity", mutate: func(value *ChunkMap) { value.RequestedChunker = " " }},
|
||||
{name: "malformed digest", mutate: func(value *ChunkMap) { value.SourceDigest = "sha256:ABC" }},
|
||||
{name: "index mismatch", mutate: func(value *ChunkMap) { value.Chunks[1].Index = 4 }},
|
||||
{name: "duplicate chunk id", mutate: func(value *ChunkMap) { value.Chunks[1].ID = value.Chunks[0].ID }},
|
||||
{name: "source mismatch", mutate: func(value *ChunkMap) { value.Chunks[0].SourceRef.SourceID = "other" }},
|
||||
{name: "invalid range", mutate: func(value *ChunkMap) { value.Chunks[0].SourceRef.StartUnitID = 0 }},
|
||||
{name: "invalid count", mutate: func(value *ChunkMap) { value.Chunks[0].UnitCount = 0 }},
|
||||
{name: "invalid namespace", mutate: func(value *ChunkMap) { value.PlanAnnotations[" "] = json.RawMessage(`null`) }},
|
||||
{name: "invalid annotation", mutate: func(value *ChunkMap) { value.Chunks[0].Annotations["test/chunker"] = json.RawMessage(`{`) }},
|
||||
{name: "plan digest mismatch", mutate: func(value *ChunkMap) { value.PlanDigest = "sha256:" + strings.Repeat("a", 64) }},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
candidate := clone(value)
|
||||
test.mutate(&candidate)
|
||||
if _, err := New().Encode(candidate); err == nil {
|
||||
t.Fatal("Encode() error = nil, want invalid durable value rejection")
|
||||
}
|
||||
})
|
||||
}
|
||||
content, err := New().Encode(value)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, raw := range [][]byte{
|
||||
append(append([]byte(nil), content[:len(content)-1]...), []byte(`,"unknown":true}`)...),
|
||||
append(append([]byte(nil), content...), []byte(` {}`)...),
|
||||
} {
|
||||
if _, err := New().Decode(raw); err == nil {
|
||||
t.Fatalf("Decode(%s) error = nil, want strict JSON rejection", raw)
|
||||
}
|
||||
}
|
||||
formatted := bytes.Replace(content, []byte(`{"label":"fixture"}`), []byte("{\n \"label\": \"fixture\"\n}"), 1)
|
||||
decoded, err := New().Decode(formatted)
|
||||
if err != nil {
|
||||
t.Fatalf("Decode(formatted annotations) error = %v", err)
|
||||
}
|
||||
if string(decoded.PlanAnnotations["test/chunker"]) != `{"label":"fixture"}` {
|
||||
t.Fatalf("decoded annotation = %s, want canonical JSON", decoded.PlanAnnotations["test/chunker"])
|
||||
}
|
||||
}
|
||||
|
||||
func TestDecodeEnforcesRequiredSchemaFieldsAndTypes(t *testing.T) {
|
||||
request := acceptedBuildRequest(t)
|
||||
request.Plan.Annotations = nil
|
||||
var err error
|
||||
request.Chunks, err = source.MaterializeChunkPlan(request.Source, request.Plan)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
artifact, err := Serialize(request)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
mutate func(map[string]any)
|
||||
}{
|
||||
{name: "missing plan annotations", mutate: func(value map[string]any) { delete(value, "plan_annotations") }},
|
||||
{name: "null plan annotations", mutate: func(value map[string]any) { value["plan_annotations"] = nil }},
|
||||
{name: "missing first index", mutate: func(value map[string]any) { delete(chunkDocument(value, 0), "index") }},
|
||||
{name: "null first index", mutate: func(value map[string]any) { chunkDocument(value, 0)["index"] = nil }},
|
||||
{name: "missing empty chunk annotations", mutate: func(value map[string]any) { delete(chunkDocument(value, 1), "annotations") }},
|
||||
{name: "null empty chunk annotations", mutate: func(value map[string]any) { chunkDocument(value, 1)["annotations"] = nil }},
|
||||
{name: "explicit empty llm profile", mutate: func(value map[string]any) {
|
||||
value["producer"].(map[string]any)["llm_profile"] = ""
|
||||
}},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
value := decodeJSONDocument(t, artifact.Content)
|
||||
test.mutate(value)
|
||||
content, err := json.Marshal(value)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := New().Decode(content); err == nil {
|
||||
t.Fatalf("Decode(%s) error = nil, want schema rejection", content)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestEncodeDoesNotMutateValue(t *testing.T) {
|
||||
value, err := Build(acceptedBuildRequest(t))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
value.Chunks[0].Annotations["test/chunker"] = json.RawMessage(" { \n \"category\" : \"sample\" \n } ")
|
||||
before := clone(value)
|
||||
if _, err := New().Encode(value); err != nil {
|
||||
t.Fatalf("Encode() error = %v", err)
|
||||
}
|
||||
if !reflect.DeepEqual(value, before) {
|
||||
t.Fatalf("Encode() mutated value:\nbefore: %#v\nafter: %#v", before, value)
|
||||
}
|
||||
}
|
||||
|
||||
func TestChunkMapOwnershipIsIndependent(t *testing.T) {
|
||||
request := acceptedBuildRequest(t)
|
||||
first, err := Build(request)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
request.Plan.Annotations["test/chunker"][0] = '['
|
||||
first.PlanAnnotations["test/chunker"][0] = '['
|
||||
second, err := Build(acceptedBuildRequest(t))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if string(second.PlanAnnotations["test/chunker"]) != `{"label":"fixture"}` {
|
||||
t.Fatalf("Build() shared mutable annotations: %s", second.PlanAnnotations["test/chunker"])
|
||||
}
|
||||
}
|
||||
|
||||
func decodeJSONDocument(t *testing.T, content []byte) map[string]any {
|
||||
t.Helper()
|
||||
decoder := json.NewDecoder(bytes.NewReader(content))
|
||||
decoder.UseNumber()
|
||||
var value map[string]any
|
||||
if err := decoder.Decode(&value); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return value
|
||||
}
|
||||
|
||||
func chunkDocument(value map[string]any, index int) map[string]any {
|
||||
return value["chunks"].([]any)[index].(map[string]any)
|
||||
}
|
||||
|
||||
func acceptedBuildRequest(t *testing.T) BuildRequest {
|
||||
t.Helper()
|
||||
document := &source.SourceDocument{
|
||||
ID: "source-test", Kind: "transcript", Format: "application/json",
|
||||
Units: []source.SourceUnit{
|
||||
{ID: 10, Kind: "segment", Text: "First unit.", Ref: source.SourceRef{SourceID: "source-test", StartUnitID: 10, EndUnitID: 10}},
|
||||
{ID: 3, Kind: "segment", Text: "Second unit.", Ref: source.SourceRef{SourceID: "source-test", StartUnitID: 3, EndUnitID: 3}},
|
||||
{ID: 20, Kind: "segment", Text: "Third unit.", Ref: source.SourceRef{SourceID: "source-test", StartUnitID: 20, EndUnitID: 20}},
|
||||
},
|
||||
}
|
||||
digest, err := source.DigestDocument(document)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
document.Digest = digest
|
||||
plan := source.ChunkPlan{
|
||||
SourceDigest: digest,
|
||||
Annotations: source.ChunkAnnotations{"test/chunker": json.RawMessage(`{"label":"fixture"}`)},
|
||||
Ranges: []source.ChunkRange{
|
||||
{StartUnitID: 10, EndUnitID: 3, Annotations: source.ChunkAnnotations{"test/chunker": json.RawMessage(`{"category":"sample"}`)}},
|
||||
{StartUnitID: 20, EndUnitID: 20},
|
||||
},
|
||||
}
|
||||
chunks, err := source.MaterializeChunkPlan(document, plan)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return BuildRequest{
|
||||
Source: document, Plan: plan, Chunks: chunks, RequestedChunker: "chunk/requested",
|
||||
Producer: Producer{InputModule: "input/producer", ChunkModule: "chunk/producer", LLMProfile: "profile/test"},
|
||||
}
|
||||
}
|
||||
51
internal/framework/chunkmap/model.go
Normal file
51
internal/framework/chunkmap/model.go
Normal file
@@ -0,0 +1,51 @@
|
||||
// Package chunkmap owns the durable accepted source chunk-map contract.
|
||||
package chunkmap
|
||||
|
||||
import (
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
)
|
||||
|
||||
const (
|
||||
ArtifactKind contracts.ArtifactKind = "source/chunk-map"
|
||||
SchemaID = "notarius.source.chunk_map"
|
||||
SchemaName = "notarius_source_chunk_map_v1"
|
||||
SchemaVersion = "v1"
|
||||
MediaType = "application/json"
|
||||
)
|
||||
|
||||
// ChunkMap is the durable representation of one accepted materialized chunk plan.
|
||||
type ChunkMap struct {
|
||||
SourceID string `json:"source_id"`
|
||||
SourceDigest string `json:"source_digest"`
|
||||
PlanDigest string `json:"plan_digest"`
|
||||
RequestedChunker string `json:"requested_chunker"`
|
||||
Producer Producer `json:"producer"`
|
||||
PlanAnnotations source.ChunkAnnotations `json:"plan_annotations"`
|
||||
Chunks []Chunk `json:"chunks"`
|
||||
}
|
||||
|
||||
// Producer identifies the component that produced the accepted logical plan.
|
||||
type Producer struct {
|
||||
InputModule string `json:"input_module"`
|
||||
ChunkModule string `json:"chunk_module"`
|
||||
LLMProfile string `json:"llm_profile,omitempty"`
|
||||
}
|
||||
|
||||
// Chunk describes one accepted materialized range without source content.
|
||||
type Chunk struct {
|
||||
ID string `json:"id"`
|
||||
Index int `json:"index"`
|
||||
SourceRef source.SourceRef `json:"source_ref"`
|
||||
UnitCount int `json:"unit_count"`
|
||||
Annotations source.ChunkAnnotations `json:"annotations"`
|
||||
}
|
||||
|
||||
// BuildRequest supplies the accepted runtime state used to build a chunk map.
|
||||
type BuildRequest struct {
|
||||
Source *source.SourceDocument
|
||||
Plan source.ChunkPlan
|
||||
Chunks []source.Chunk
|
||||
RequestedChunker string
|
||||
Producer Producer
|
||||
}
|
||||
1
internal/framework/chunkmap/testdata/source_chunk_map.v1.json
vendored
Normal file
1
internal/framework/chunkmap/testdata/source_chunk_map.v1.json
vendored
Normal file
@@ -0,0 +1 @@
|
||||
{"source_id":"source-test","source_digest":"sha256:186b2d30029e7fda40545f88e75ade38f22bff3e5543af4dfeb86587536a01af","plan_digest":"sha256:e50cc7da9070ac8a4339d79b2c206be842846c5c22af7b0ad58f7e9a34ceaf97","requested_chunker":"chunk/requested","producer":{"input_module":"input/producer","chunk_module":"chunk/producer","llm_profile":"profile/test"},"plan_annotations":{"test/chunker":{"label":"fixture"}},"chunks":[{"id":"chunk-000001","index":0,"source_ref":{"source_id":"source-test","start_unit_id":10,"end_unit_id":3},"unit_count":2,"annotations":{"test/chunker":{"category":"sample"}}},{"id":"chunk-000002","index":1,"source_ref":{"source_id":"source-test","start_unit_id":20,"end_unit_id":20},"unit_count":1,"annotations":{}}]}
|
||||
@@ -281,7 +281,7 @@ func TestFilesystemStoreReportsInvalidRecordsAsRecoverable(t *testing.T) {
|
||||
return bytes.Replace(data, []byte(`{"schema_version"`), []byte(`{"SENTINEL_UNKNOWN_FIELD":true,"schema_version"`), 1)
|
||||
}},
|
||||
{name: "truncated JSON", mutate: func(data []byte) []byte { return data[:len(data)/2] }},
|
||||
{name: "schema mismatch", mutate: replaceJSON(`notarius.chunk-plan.v1`, `SENTINEL_SCHEMA_VALUE`)},
|
||||
{name: "legacy v1 record", mutate: replaceJSON(`notarius.chunk-plan.v2`, `notarius.chunk-plan.v1`)},
|
||||
{name: "source mismatch", mutate: replaceJSON(testSourceDigest, "sha256:"+strings.Repeat("b", 64))},
|
||||
{name: "plan digest mismatch", mutate: func(data []byte) []byte {
|
||||
prefix := []byte(`"plan_digest":"sha256:`)
|
||||
|
||||
@@ -69,6 +69,16 @@ func CloneSerializedArtifact(artifact SerializedArtifact) SerializedArtifact {
|
||||
return artifact
|
||||
}
|
||||
|
||||
// CloneSerializedArtifactPointer returns an independently owned artifact when
|
||||
// one is present.
|
||||
func CloneSerializedArtifactPointer(artifact *SerializedArtifact) *SerializedArtifact {
|
||||
if artifact == nil {
|
||||
return nil
|
||||
}
|
||||
cloned := CloneSerializedArtifact(*artifact)
|
||||
return &cloned
|
||||
}
|
||||
|
||||
func CloneSerializedOutput(output SerializedOutput) SerializedOutput {
|
||||
output.Artifact = CloneSerializedArtifact(output.Artifact)
|
||||
return output
|
||||
|
||||
@@ -43,6 +43,7 @@ type LLMDebugPrompt struct {
|
||||
PromptVersion string `json:"prompt_version,omitempty"`
|
||||
PromptHash string `json:"prompt_hash,omitempty"`
|
||||
SelectedProfileID string `json:"selected_profile_id,omitempty"`
|
||||
SelectedBackendID string `json:"selected_backend_id,omitempty"`
|
||||
SessionID string `json:"session_id,omitempty"`
|
||||
RenderedPromptHash string `json:"rendered_prompt_hash,omitempty"`
|
||||
Messages []LLMDebugMessage `json:"messages,omitempty"`
|
||||
@@ -266,10 +267,6 @@ const (
|
||||
ExecutionClassLLMBacked ExecutionClass = "llm_backed"
|
||||
)
|
||||
|
||||
type ChunkExecutionClassProvider interface {
|
||||
ExecutionClass() ExecutionClass
|
||||
}
|
||||
|
||||
type ValidationResult struct {
|
||||
Approved bool `json:"approved"`
|
||||
ReasonCode string `json:"reason_code,omitempty"`
|
||||
@@ -291,6 +288,8 @@ type OutputRequest struct {
|
||||
Warnings []Warning `json:"warnings,omitempty"`
|
||||
LLMProfile string `json:"llm_profile,omitempty"`
|
||||
Metadata map[string]any `json:"metadata,omitempty"`
|
||||
ChunkMap *SerializedArtifact `json:"chunk_map,omitempty"`
|
||||
EvidenceContext *SerializedArtifact `json:"evidence_context,omitempty"`
|
||||
}
|
||||
|
||||
type OutputFile struct {
|
||||
|
||||
@@ -26,6 +26,25 @@ func TestCloneReferenceSlotsEmptyInputReturnsNil(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestCloneSerializedArtifactPointerOwnsArtifactData(t *testing.T) {
|
||||
original := &SerializedArtifact{
|
||||
Kind: "test/artifact",
|
||||
Schema: ArtifactSchema{ID: "test.schema", JSONSchema: []byte(`{"type":"object"}`)},
|
||||
Content: []byte(`{"items":[]}`),
|
||||
Metadata: map[string]any{"count": 1},
|
||||
}
|
||||
cloned := CloneSerializedArtifactPointer(original)
|
||||
original.Schema.JSONSchema[0] = '['
|
||||
original.Content[0] = '['
|
||||
original.Metadata["count"] = 2
|
||||
if cloned == nil || string(cloned.Schema.JSONSchema) != `{"type":"object"}` || string(cloned.Content) != `{"items":[]}` || cloned.Metadata["count"] != 1 {
|
||||
t.Fatalf("CloneSerializedArtifactPointer() = %#v, want independent artifact data", cloned)
|
||||
}
|
||||
if CloneSerializedArtifactPointer(nil) != nil {
|
||||
t.Fatal("CloneSerializedArtifactPointer(nil) must return nil")
|
||||
}
|
||||
}
|
||||
|
||||
func TestCloneReferenceSlotsPreservesFields(t *testing.T) {
|
||||
slots := []ReferenceSlot{
|
||||
{
|
||||
|
||||
11
internal/framework/contracts/errors.go
Normal file
11
internal/framework/contracts/errors.go
Normal file
@@ -0,0 +1,11 @@
|
||||
package contracts
|
||||
|
||||
import "errors"
|
||||
|
||||
// ErrInvalidStructuredOutput identifies a provider response that cannot satisfy
|
||||
// the caller's declared structured-output contract.
|
||||
var ErrInvalidStructuredOutput = errors.New("invalid structured output")
|
||||
|
||||
// ErrLLMCapacityExceeded identifies backend admission exhaustion before model
|
||||
// generation begins.
|
||||
var ErrLLMCapacityExceeded = errors.New("LLM capacity exceeded")
|
||||
@@ -90,6 +90,22 @@ type TypedNormalizeRequest[T any] struct {
|
||||
type TypedNormalizeResult[T any] struct {
|
||||
Value T
|
||||
Warnings []Warning
|
||||
Retry *NormalizeRetry
|
||||
}
|
||||
|
||||
// Normalize retry diagnostic limits bound module-provided values before the
|
||||
// framework persists them in debug artifacts.
|
||||
const (
|
||||
MaxNormalizeRetryReasonCodeBytes = 128
|
||||
MaxNormalizeRetryMessageBytes = 4096
|
||||
)
|
||||
|
||||
// NormalizeRetry asks the framework to retry normalization while retaining a
|
||||
// safe candidate for acceptance if the retry budget is exhausted.
|
||||
type NormalizeRetry struct {
|
||||
ReasonCode string
|
||||
Message string
|
||||
FallbackWarnings []Warning
|
||||
}
|
||||
|
||||
type Normalizer[T any] interface {
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user