Compare commits
348 Commits
v0.1.0
...
987c9691c6
| Author | SHA1 | Date | |
|---|---|---|---|
| 987c9691c6 | |||
| 9e4b989e53 | |||
| 6bdebcb5e2 | |||
| ad36d534a4 | |||
| 73d6e184ab | |||
| 64d461fc18 | |||
| fb1134e591 | |||
| 1da29e6788 | |||
| 0d947549fb | |||
| 950fba17ce | |||
| 678d2c6099 | |||
| db8db5ffc5 | |||
| 94b3eafb1a | |||
| f6981e2264 | |||
| 2f506f4985 | |||
| fdf8c4afd4 | |||
| 74c793e6a1 | |||
| b5835fbc37 | |||
| fd3f7b85cc | |||
| d86b74f485 | |||
| 59cbf1eb27 | |||
| ee43add75c | |||
| 46761706a2 | |||
| 5a968b64eb | |||
| 7b077c269d | |||
| 80ec939383 | |||
| d3c4d6f133 | |||
| 5ad661f95f | |||
| d63e5c6852 | |||
| fbb8e0d241 | |||
| 8d9a496935 | |||
| d1c48db4bc | |||
| 6bd781d344 | |||
| 8a12c56971 | |||
| 26bd59a5a2 | |||
| 927a7beb88 | |||
| f7059607af | |||
| 63de44c347 | |||
| f0ede9dacc | |||
| 5711f8b9e3 | |||
| da83510234 | |||
| f51b22bea7 | |||
| f320c2fcee | |||
| 4ba1e50a89 | |||
| 2a7e025251 | |||
| 3da20e9d6a | |||
| b1c0faa748 | |||
| 989f2c220b | |||
| 7e35915b3e | |||
| 24238d249e | |||
| 9614469b45 | |||
| 29ee68824d | |||
| aeaaf44ae0 | |||
| d752c51aec | |||
| 8199d95dc1 | |||
| 97cdb01357 | |||
| 7a66095912 | |||
| 9d1356a20e | |||
| e4471fc300 | |||
| 84a2854b5e | |||
| a1b76093ce | |||
| 1aa30a73db | |||
| dc7c0e2f9e | |||
| e2cb0d901a | |||
| 8e0b029f5f | |||
| 9bbf2535dd | |||
| 83fde83a58 | |||
| bef3d1359d | |||
| 1ff449435f | |||
| 6e21c83fd8 | |||
| 6dc9d522b1 | |||
| 8adcf6840d | |||
| f5ed30e455 | |||
| cacf3f24e7 | |||
| f08ca4ddfa | |||
| e1c2f3c202 | |||
| 9614eb540d | |||
| 1b46596a39 | |||
| e043d61a99 | |||
| ad89782c9b | |||
| cd29265d5d | |||
| 2f36b7c3b6 | |||
| b8b3f3abfa | |||
| b490297cde | |||
| 06148074a2 | |||
| 16a998055c | |||
| 97c9a8e5ce | |||
| 66415fd1fa | |||
| bfe25609a7 | |||
| 36e0512454 | |||
| b02f667107 | |||
| ed2b6f4580 | |||
| cb7f145c76 | |||
| 2b9d2eaeaa | |||
| 61016671ab | |||
| b2c076946b | |||
| 250c5c22b8 | |||
| 4b0b166143 | |||
| 90481a0e4b | |||
| 2cbaf20e55 | |||
| a263a0840c | |||
| 14991cf58b | |||
| ab0b4e350c | |||
| 7b2fb0880d | |||
| 748e02db80 | |||
| 23c55f8925 | |||
| 906d97b391 | |||
| f15fd4f9c1 | |||
| 7aadb088a6 | |||
| 9de399432e | |||
| 5bdd56cfb1 | |||
| 64ea23c21f | |||
| 7071102ab7 | |||
| 9184072839 | |||
| c437682407 | |||
| 22d4f29670 | |||
| afb7ed3cf1 | |||
| f846f252c0 | |||
| f5618d1f0c | |||
| f94ab0a6bf | |||
| 41b52aae74 | |||
| ed36f7d7fd | |||
| 3ba2bfd7f6 | |||
| b344d16dc1 | |||
| 6e12c09952 | |||
| 07460341e3 | |||
| 732b13669f | |||
| 3a8a82ebc9 | |||
| 1c9819f08e | |||
| 447c4f73f9 | |||
| e01b8d1b6d | |||
| 3d70920f3d | |||
| 110593ece1 | |||
| a1f5dce405 | |||
| 50aa60e0b8 | |||
| 2fbb3813aa | |||
| 6dd695611c | |||
| 92acb45775 | |||
| d3e171aa82 | |||
| c6f330eb06 | |||
| 20cfbfd311 | |||
| fb043325e1 | |||
| 06c0259788 | |||
| 3d5fd9dc05 | |||
| 3ba2c62cc1 | |||
| e2ab01f9d2 | |||
| 5186e061a8 | |||
| c4c907d421 | |||
| fa5076f5f1 | |||
| 8b5a4e0efd | |||
| 3eb68baca6 | |||
| ae97adb8b0 | |||
| 79b9fffcaf | |||
| be22852daa | |||
| f5107045c3 | |||
| 2c98763b9b | |||
| d2eb763b9b | |||
| 0f25e7339f | |||
| 87c57681f6 | |||
| 3d0d79360e | |||
| f08b407b72 | |||
| 4ff2c7795f | |||
| 3bfe05ab56 | |||
| 7806dba509 | |||
| ac53f83ac8 | |||
| 385e4593f4 | |||
| f64bb7c883 | |||
| 9d3175d36a | |||
| 4f96abf42c | |||
| d88bcb6070 | |||
| 0cca3b1f5d | |||
| bbc83ab042 | |||
| 2cba6d4512 | |||
| e70450c401 | |||
| 4d3351c774 | |||
| a586257d5e | |||
| b8163091cc | |||
| c7b3af82b4 | |||
| a42b06ba20 | |||
| 8d62973627 | |||
| 8cdefc72a1 | |||
| d3a8dc7930 | |||
| 86ebb62f84 | |||
| 50191ee694 | |||
| 3c35124db4 | |||
| e4ec521bed | |||
| 2111e01142 | |||
| a39eea7ed6 | |||
| 7bcce9953e | |||
| 9746a42e04 | |||
| 47bacc7abb | |||
| 8824948910 | |||
| 8cb11e60e4 | |||
| 26142f0e05 | |||
| 1542a12497 | |||
| a9250206d5 | |||
| 5bd0ba7a72 | |||
| 286fb9dce7 | |||
| 9fa9154dda | |||
| 604c7a7945 | |||
| a0f5e6e2b9 | |||
| 561d65a505 | |||
| 205e2a9908 | |||
| 8c59b6af14 | |||
| b3328b93e5 | |||
| 96a49bb7cd | |||
| 6d0a19c94c | |||
| 6fc6ce0adb | |||
| 8ba5228c01 | |||
| 51a36efb6b | |||
| ebd449d847 | |||
| 7844c0a93f | |||
| 3bfac14397 | |||
| 1c13e1d64a | |||
| 60b86dc40c | |||
| 68481804a7 | |||
| 236ccc62ad | |||
| 35bffdf336 | |||
| 3772b308e9 | |||
| ef6926322d | |||
| 2df7084d5d | |||
| 3d3cc0c08e | |||
| fbc3d9add6 | |||
| 35e45f0914 | |||
| 3e4fa923eb | |||
| 3013ee044d | |||
| adfd3bd052 | |||
| 4023c66508 | |||
| adfe3825ee | |||
| 814fcdc6ba | |||
| 66de1a5520 | |||
| 52e6b31408 | |||
| 142ba36695 | |||
| b949e9bbc0 | |||
| ce3a07512f | |||
| 1c84d19e5f | |||
| fc1b57bde2 | |||
| 075888c97f | |||
| 40709e4ad8 | |||
| 15c369c509 | |||
| a81b9f1e1f | |||
| 0327659355 | |||
| c99bad19ae | |||
| 35f9446ed8 | |||
| 21888d625f | |||
| feb03c3f8e | |||
| 3b07b64a0f | |||
| 6e6375521d | |||
| b1fe9dc5a7 | |||
| 6db2dc8d2a | |||
| 98b03a4629 | |||
| 610bdb4fea | |||
| 68ec69f2e4 | |||
| 451f6c0bb9 | |||
| 3011dd91ca | |||
| ae65b95374 | |||
| a5bbfea9b9 | |||
| ae9c2e1d5e | |||
| 1d3a444df8 | |||
| f044c00a7c | |||
| 7d89c2702b | |||
| 93653cccb8 | |||
| a024492dbf | |||
| 304c68f9fc | |||
| c5f2b14ff4 | |||
| fc8e03f98c | |||
| 16de4b6437 | |||
| 0f30888b00 | |||
| 3e67be6ac3 | |||
| 5ef027b6f0 | |||
| 666b4bf801 | |||
| d593bfee0a | |||
| b7ad66f0e0 | |||
| 249e49c928 | |||
| e54e74ed88 | |||
| 582c5dceed | |||
| a9d8505cdb | |||
| 7c95791e94 | |||
| aa14faa3cb | |||
| cc6b050367 | |||
| bcedf19a08 | |||
| c05ecb58d8 | |||
| 9e3f8809b3 | |||
| 4f057b99ac | |||
| aec807fcb0 | |||
| 79a585d17e | |||
| 671ff6d132 | |||
| 9b2d0297b7 | |||
| f91e643932 | |||
| 35fe405448 | |||
| 524f2ffb8e | |||
| 68b426cdb0 | |||
| 223f3751e8 | |||
| 7861d040df | |||
| 47cf7e76ec | |||
| 8cafa64174 | |||
| ecba0ad725 | |||
| b3757dcf7b | |||
| 3e456ec4d4 | |||
| 5b1efc89f6 | |||
| aee48d011e | |||
| 3217bb3e12 | |||
| 3df686f474 | |||
| 7d4c027d09 | |||
| 31d70a2dd7 | |||
| c9fbb331e2 | |||
| f6224dcbee | |||
| de6689bc1d | |||
| 0fc740470f | |||
| 49d94cc2e9 | |||
| 291298cf7b | |||
| 9532ae8121 | |||
| 7601731a2c | |||
| 3aa88ab9d3 | |||
| c1ba94192d | |||
| 22032dfd6d | |||
| 4cafde2502 | |||
| 8c623b7ad8 | |||
| 43dc954440 | |||
| 51053d390d | |||
| 9278797aa9 | |||
| 39e49d7f77 | |||
| 84c4c06712 | |||
| eab640aa21 | |||
| a516944086 | |||
| be6803ffa1 | |||
| ef4bdd4f9f | |||
| 2f97895732 | |||
| 9e89b88efc | |||
| a57c6397e3 | |||
| 39e071f5ca | |||
| 70d733edaf | |||
| 1c31f56af1 | |||
| f9999a73df | |||
| 11d8187052 | |||
| 86bff552c1 | |||
| d3f790095e | |||
| 95218218e2 | |||
| e700df82d8 | |||
| e19cc02c4d | |||
| 8a5419448f | |||
| c8217549a8 | |||
| 2130414899 | |||
| 7f83a20fa6 | |||
| 317ab0472d | |||
| e5eb0ba5c8 | |||
| b95af4f87d | |||
| 11073b613c |
3
.codebase-memory/.gitattributes
vendored
Normal file
3
.codebase-memory/.gitattributes
vendored
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
# Auto-generated by codebase-memory-mcp
|
||||||
|
# Prevent merge conflicts on compressed artifact
|
||||||
|
graph.db.zst merge=ours binary
|
||||||
11
.codebase-memory/artifact.json
Normal file
11
.codebase-memory/artifact.json
Normal file
@@ -0,0 +1,11 @@
|
|||||||
|
{
|
||||||
|
"schema_version": 2,
|
||||||
|
"commit": "9e4b989e53d65efa614b5eedd8530caed12f60b5",
|
||||||
|
"indexed_at": "2026-07-27T18:27:55Z",
|
||||||
|
"project": "home-eric-Workspace-notarius",
|
||||||
|
"nodes": 6356,
|
||||||
|
"edges": 35132,
|
||||||
|
"original_size": 26017792,
|
||||||
|
"compressed_size": 4386397,
|
||||||
|
"compression_level": 3
|
||||||
|
}
|
||||||
BIN
.codebase-memory/graph.db.zst
Normal file
BIN
.codebase-memory/graph.db.zst
Normal file
Binary file not shown.
6
.gitignore
vendored
6
.gitignore
vendored
@@ -1,3 +1,8 @@
|
|||||||
|
# build and testing artifacts
|
||||||
|
notarius
|
||||||
|
notarius-output
|
||||||
|
workspace/
|
||||||
|
|
||||||
# ---> Go
|
# ---> Go
|
||||||
# If you prefer the allow list template instead of the deny list, see community template:
|
# If you prefer the allow list template instead of the deny list, see community template:
|
||||||
# https://github.com/github/gitignore/blob/main/community/Golang/Go.AllowList.gitignore
|
# https://github.com/github/gitignore/blob/main/community/Golang/Go.AllowList.gitignore
|
||||||
@@ -49,6 +54,7 @@ go.work.sum
|
|||||||
# Icon must end with two \r
|
# Icon must end with two \r
|
||||||
Icon
|
Icon
|
||||||
|
|
||||||
|
|
||||||
# Thumbnails
|
# Thumbnails
|
||||||
._*
|
._*
|
||||||
|
|
||||||
|
|||||||
@@ -1,3 +1,2 @@
|
|||||||
Please carefully review the documents in `docs/policy` before making any changes to this repository.
|
Please review `docs/development.md` for initial orientation in this repository
|
||||||
- `architecture.md` provides the canonical high-level architecture policy for this repository.
|
and follow its task-specific reading guide.
|
||||||
- `documentation.md` provides the canonical documentation policy for this repository.
|
|
||||||
|
|||||||
63
README.md
63
README.md
@@ -1,36 +1,45 @@
|
|||||||
# Notarius
|
# Notarius
|
||||||
|
|
||||||
Notarius is a Go CLI for extracting structured artifacts from source material
|
Notarius is a Go CLI for turning source material into structured artifacts with
|
||||||
with explicit, configurable pipeline modules.
|
configured extraction pipelines. The implemented D&D workflow reads Seriatim
|
||||||
|
transcript JSON and can produce scene descriptions, item and currency events,
|
||||||
|
NPC identities, combat turns, NPC interactions, and spell casts.
|
||||||
|
|
||||||
The current implementation reads Seriatim transcript JSON, chunks the source
|
## Quickstart
|
||||||
units, extracts D&D spell-cast artifacts with an OpenAI-compatible LLM, and
|
|
||||||
writes JSON output plus diagnostics for each run.
|
|
||||||
|
|
||||||
```sh
|
Provide an OpenRouter API key through the environment, then run the maintained
|
||||||
NOTARIUS_LLM_DEFAULT_BASE_URL=http://127.0.0.1:8080/v1 \
|
minimal example:
|
||||||
NOTARIUS_LLM_DEFAULT_MODEL=your-model \
|
|
||||||
|
~~~
|
||||||
|
OPENROUTER_API_KEY=your-api-key \
|
||||||
go run ./cmd/notarius run dnd-session \
|
go run ./cmd/notarius run dnd-session \
|
||||||
--config examples/dnd-spells.config.yml \
|
--config examples/dnd-minimal.config.yml \
|
||||||
--input examples/seriatim-minimal-transcript.json
|
--input examples/seriatim-minimal-transcript.json
|
||||||
```
|
~~~
|
||||||
|
|
||||||
If the provider requires authentication, set
|
The command publishes a JSON output bundle. Its command syntax and exit
|
||||||
`NOTARIUS_LLM_DEFAULT_API_KEY` in the environment before running the command.
|
behavior are documented in the [CLI reference](docs/cli.md); configuration,
|
||||||
Outputs are written under `./notarius-output/<run-id>/` unless `--output-dir`
|
credentials, and module selection are owned by the
|
||||||
is provided.
|
[configuration reference](docs/config.md).
|
||||||
|
|
||||||
Useful references:
|
For the complete ordered D&D workflow, use
|
||||||
|
[the complete configuration](examples/dnd-complete.config.yml) with
|
||||||
|
[its synthetic transcript](examples/dnd-complete-transcript.json). It
|
||||||
|
demonstrates all implemented D&D lanes and the supporting campaign references.
|
||||||
|
|
||||||
- [CLI reference](docs/cli.md)
|
## Documentation
|
||||||
- [Configuration reference](docs/config.md)
|
|
||||||
- [Operations](docs/operations.md)
|
- [CLI reference](docs/cli.md) — commands, flags, output streams, and exits.
|
||||||
- [Troubleshooting](docs/troubleshooting.md)
|
- [Configuration reference](docs/config.md) — configuration files, profiles,
|
||||||
- [Seriatim input contract](docs/integrations/seriatim.md)
|
validation, and module selection.
|
||||||
- [OpenAI-compatible provider contract](docs/integrations/openai-compatible.md)
|
- [Operations](docs/operations.md) — output, state, recovery, and debug
|
||||||
- [JSON output contract](docs/integrations/json-output.md)
|
handling.
|
||||||
- [D&D spell artifact contract](docs/integrations/dnd-spell-artifacts.md)
|
- [Integration contracts](docs/integrations/) — Seriatim input and published
|
||||||
- [Developer workflow](docs/policy/development.md)
|
artifact formats.
|
||||||
- [Internal architecture docs](docs/internal/overview.md)
|
- [Subprocess consumer guide](docs/consumers/subprocess.md) — invoke Notarius
|
||||||
- [Maintained example config](examples/dnd-spells.config.yml)
|
from an orchestrator and consume a published result.
|
||||||
- [Maintained example input](examples/seriatim-minimal-transcript.json)
|
- [Internal overview](docs/internal/overview.md) — implemented component map
|
||||||
|
for maintainers.
|
||||||
|
- [Developer guide](docs/development.md) — contributor orientation and
|
||||||
|
validation guidance.
|
||||||
|
- [Future work](docs/roadmap/future.md) — unimplemented ideas and priorities.
|
||||||
|
|||||||
23
docs/adr/0001-record-architecture-decisions.md
Normal file
23
docs/adr/0001-record-architecture-decisions.md
Normal file
@@ -0,0 +1,23 @@
|
|||||||
|
# ADR-0001: Record architecture decisions as ADRs
|
||||||
|
|
||||||
|
**Status:** Accepted
|
||||||
|
**Date:** 2026-07-13
|
||||||
|
|
||||||
|
## Context
|
||||||
|
Architectural reasoning made during design (pattern choices, rejected
|
||||||
|
alternatives, trigger conditions for revisiting) is lost if only the final
|
||||||
|
state is documented.
|
||||||
|
|
||||||
|
## Decision
|
||||||
|
We keep a living overview in docs/policy/architecture.md describing current
|
||||||
|
intended state, and immutable, numbered ADRs (Nygard format) in docs/adr/
|
||||||
|
recording each significant decision, its alternatives, and its consequences.
|
||||||
|
Changed decisions get a new ADR that marks the old one Superseded.
|
||||||
|
|
||||||
|
## Alternatives considered
|
||||||
|
- Overview doc only: loses the "why" and the rejected options.
|
||||||
|
- arc42 / RFC-style design docs: heavier than warranted for a solo repo.
|
||||||
|
|
||||||
|
## Consequences
|
||||||
|
Small ongoing writing cost; durable reasoning trail; cheap onboarding for
|
||||||
|
future contributors (including future-us).
|
||||||
51
docs/adr/0002-linear-pipes-and-filters-pipeline.md
Normal file
51
docs/adr/0002-linear-pipes-and-filters-pipeline.md
Normal file
@@ -0,0 +1,51 @@
|
|||||||
|
# ADR-0002: Linear pipes-and-filters pipeline, not a general DAG
|
||||||
|
|
||||||
|
**Status:** Accepted
|
||||||
|
**Date:** 2026-07-13
|
||||||
|
|
||||||
|
## Context
|
||||||
|
|
||||||
|
Notarius processes source material through one known workflow:
|
||||||
|
|
||||||
|
```text
|
||||||
|
input -> chunk -> extract -> merge -> normalize -> output
|
||||||
|
```
|
||||||
|
|
||||||
|
Input and chunking apply to the source as a whole. Each selected artifact lane
|
||||||
|
then performs extract, merge, and normalize, after which output aggregates the
|
||||||
|
lane outcomes. Chunk extraction has a natural scatter-gather shape, but no
|
||||||
|
current use case requires arbitrary branches, joins, or user-defined stage
|
||||||
|
topology.
|
||||||
|
|
||||||
|
## Decision
|
||||||
|
|
||||||
|
Notarius implements a fixed six-stage pipes-and-filters pipeline. Configuration
|
||||||
|
selects implementations for these stages but cannot add stages, reorder them,
|
||||||
|
or define an arbitrary graph.
|
||||||
|
|
||||||
|
The framework owns stage sequencing and the scatter-gather boundary between
|
||||||
|
chunk, extract, and merge. Extract results are handed to merge in deterministic
|
||||||
|
source-chunk order regardless of execution strategy. Each artifact lane remains
|
||||||
|
logically linear. Output runs after every selected lane has either produced an
|
||||||
|
accepted normalized artifact or reached a recorded rejection. A framework
|
||||||
|
execution failure aborts the pipeline.
|
||||||
|
|
||||||
|
The runner's concrete internal representation and stage-specific scheduling
|
||||||
|
policies are implementation details. Concurrency must preserve the pipeline's
|
||||||
|
deterministic handoffs, validation behavior, and provenance, and all execution
|
||||||
|
strategies must continue to honor context cancellation.
|
||||||
|
|
||||||
|
## Alternatives considered
|
||||||
|
|
||||||
|
- Build a general DAG engine now. This would support hypothetical branching
|
||||||
|
topologies, but would add scheduling, topology validation, configuration, and
|
||||||
|
state-management complexity without a current consumer. Revisit this choice
|
||||||
|
only when a concrete workflow requires a topology the fixed pipeline cannot
|
||||||
|
express.
|
||||||
|
|
||||||
|
## Consequences
|
||||||
|
|
||||||
|
The runner, configuration model, and operator mental model remain small. Stage
|
||||||
|
ownership stays visible, and general chunking, merging, or normalization cannot
|
||||||
|
be hidden inside extractors. A future DAG requirement will require an explicit
|
||||||
|
architectural change rather than incremental exceptions to the fixed pipeline.
|
||||||
119
docs/adr/0003-typed-interfaces-with-two-zone-data-model.md
Normal file
119
docs/adr/0003-typed-interfaces-with-two-zone-data-model.md
Normal file
@@ -0,0 +1,119 @@
|
|||||||
|
# ADR-0003: Strongly typed stage interfaces with a two-zone data model
|
||||||
|
|
||||||
|
**Status:** Accepted
|
||||||
|
**Date:** 2026-07-13
|
||||||
|
|
||||||
|
## Context
|
||||||
|
|
||||||
|
Pipeline stages must exchange source data and extracted artifacts. Universal
|
||||||
|
source data has one engine-wide meaning, while extracted artifacts have
|
||||||
|
domain-specific shapes. Passing opaque bytes or `any` between all stages would
|
||||||
|
make invalid wiring and merge behavior runtime concerns. Requiring JSON at
|
||||||
|
every handoff would preserve interoperability but discard useful Go type safety
|
||||||
|
while all modules are in-process.
|
||||||
|
|
||||||
|
The framework must also support multiple configured artifact domains, durable
|
||||||
|
checkpoints, diagnostics, and output encoders without making those consumers
|
||||||
|
depend on every domain's Go types.
|
||||||
|
|
||||||
|
## Decision
|
||||||
|
|
||||||
|
Notarius uses two typed data zones followed by one serialized boundary.
|
||||||
|
|
||||||
|
### Source zone
|
||||||
|
|
||||||
|
Input and chunk stages use conservative, engine-owned document, segment, chunk,
|
||||||
|
and source-reference types. Their exact Go names are implementation details.
|
||||||
|
Every segment carries engine-owned source provenance identifying the source
|
||||||
|
location from which it was produced. Chunks preserve the ordered provenance of
|
||||||
|
their segments.
|
||||||
|
|
||||||
|
Source-format-specific fields remain in input modules or explicitly namespaced
|
||||||
|
metadata; they do not become framework contracts.
|
||||||
|
|
||||||
|
### Domain artifact zone
|
||||||
|
|
||||||
|
Each artifact lane has one domain-owned Go artifact type `T`. Its extract,
|
||||||
|
merge, normalize, and domain-aware validation implementations use generic,
|
||||||
|
strongly typed contracts over the same `T`. Raw JSON, opaque bytes, and `any`
|
||||||
|
are not stage-handoff contracts within a lane.
|
||||||
|
|
||||||
|
Each registered domain artifact type supplies a codec for `T`. The codec owns:
|
||||||
|
|
||||||
|
- stable schema identity and an explicit schema version;
|
||||||
|
- JSON serialization and deserialization;
|
||||||
|
- the media type and schema metadata required at serialized boundaries; and
|
||||||
|
- rejection of data that cannot be represented by the declared artifact
|
||||||
|
schema.
|
||||||
|
|
||||||
|
An artifact type's JSON representation is a maintained domain contract.
|
||||||
|
Changing it incompatibly requires a new schema version.
|
||||||
|
|
||||||
|
Extract, merge, and normalize may change the contents of `T`, but they do not
|
||||||
|
change the lane's canonical Go artifact type or artifact schema identity. An
|
||||||
|
extractor maps any provider- or prompt-specific response type into `T` before
|
||||||
|
returning. A future lane that requires different artifact types at different
|
||||||
|
stages requires a new architectural decision.
|
||||||
|
|
||||||
|
### Serialized boundary
|
||||||
|
|
||||||
|
After normalization, each typed artifact is converted into an engine-owned
|
||||||
|
serialized artifact containing bytes, media type, and schema metadata. Output
|
||||||
|
aggregation and output encoders consume this type-erased form. Intermediate
|
||||||
|
checkpoint and debug encodings do not become stage-handoff contracts.
|
||||||
|
|
||||||
|
LLM transport, checkpoints, and opt-in debug recording are also explicit
|
||||||
|
serialization boundaries. They may encode or decode a typed artifact through
|
||||||
|
its domain codec, but they do not change the in-memory type used between
|
||||||
|
extract, merge, normalize, and typed validators. Checkpoint reuse requires a
|
||||||
|
compatible schema identity and version.
|
||||||
|
|
||||||
|
An LLM structured-response schema is a module transport contract and may differ
|
||||||
|
from the domain artifact schema. The calling module owns the response type and
|
||||||
|
maps it into the canonical `T`; the artifact codec remains authoritative for
|
||||||
|
artifact checkpoints and output serialization.
|
||||||
|
|
||||||
|
The framework may use private type-erased adapters to store heterogeneous lane
|
||||||
|
registrations and execute configured domains. Such an adapter must assemble a
|
||||||
|
type-consistent lane before execution and must not expose `any` or raw payloads
|
||||||
|
as module-facing handoffs inside the domain artifact zone.
|
||||||
|
|
||||||
|
### Construction and dependencies
|
||||||
|
|
||||||
|
Every module operation accepts `context.Context`. Modules receive stable runtime
|
||||||
|
collaborators through an injected dependency set at construction time. In
|
||||||
|
particular, LLM-using modules receive the application-provided structured LLM
|
||||||
|
client and do not construct provider clients or bypass shared scheduling.
|
||||||
|
|
||||||
|
The application boundary enforces one configurable global upper bound on
|
||||||
|
in-flight LLM calls across all stages, lanes, retries, and validators.
|
||||||
|
|
||||||
|
Configuration options are parsed and validated while a module is constructed,
|
||||||
|
before that module executes. Per-run data such as source material, references,
|
||||||
|
session identity, and lane identity remains operation input rather than a
|
||||||
|
construction dependency.
|
||||||
|
|
||||||
|
## Alternatives considered
|
||||||
|
|
||||||
|
- Pass raw bytes between stages. This maximizes decoupling but moves wiring,
|
||||||
|
parsing, and merge errors to runtime and prevents domain types from being the
|
||||||
|
canonical in-process contract.
|
||||||
|
- Require JSON plus schemas at every stage boundary. This is appropriate for an
|
||||||
|
out-of-process boundary, but adds serialization and parsing inside the current
|
||||||
|
in-process pipeline. The stable codec contract preserves this upgrade path if
|
||||||
|
remote plugins are introduced.
|
||||||
|
- Use a uniform `Process(any) (any, error)` contract. This simplifies a fully
|
||||||
|
dynamic engine but turns incompatible module composition into type assertions
|
||||||
|
and runtime failures. The fixed topology does not require that tradeoff.
|
||||||
|
|
||||||
|
## Consequences
|
||||||
|
|
||||||
|
Domain pipelines gain compile-time handoff safety and explicit merge semantics.
|
||||||
|
Serialization, schema compatibility, checkpoint decoding, and output erasure
|
||||||
|
have named owners. Dynamic registration requires a small erased adapter around
|
||||||
|
each typed lane, and generic stage implementations must be instantiated for a
|
||||||
|
specific artifact type or behavior rather than manipulating arbitrary JSON.
|
||||||
|
|
||||||
|
The engine-owned source model becomes a long-lived contract and must evolve
|
||||||
|
conservatively. Domain authors must maintain a codec and versioned schema in
|
||||||
|
addition to their Go artifact type.
|
||||||
81
docs/adr/0004-package-modules-by-domain.md
Normal file
81
docs/adr/0004-package-modules-by-domain.md
Normal file
@@ -0,0 +1,81 @@
|
|||||||
|
# ADR-0004: Package modules by domain, not by stage
|
||||||
|
|
||||||
|
**Status:** Accepted
|
||||||
|
**Date:** 2026-07-13
|
||||||
|
|
||||||
|
## Context
|
||||||
|
|
||||||
|
Module packages can be grouped first by pipeline stage, such as
|
||||||
|
`modules/chunk/dnd/scenes`, or first by domain, such as
|
||||||
|
`modules/dnd/chunk/scenes`. A domain's extract, merge, normalize, validation,
|
||||||
|
schema, prompt, and artifact-codec implementations collaborate around the same
|
||||||
|
artifact types and are likely to evolve together.
|
||||||
|
|
||||||
|
Go package dependencies also constrain registration. If shared types live in a
|
||||||
|
domain root package, that package cannot import child implementation packages
|
||||||
|
to register them because the children already import the root types.
|
||||||
|
|
||||||
|
## Decision
|
||||||
|
|
||||||
|
Production extensions are grouped by domain under:
|
||||||
|
|
||||||
|
```text
|
||||||
|
internal/modules/<domain>/<stage>/<name>
|
||||||
|
```
|
||||||
|
|
||||||
|
Shared artifact types live at the domain root, for example
|
||||||
|
`internal/modules/dnd/types.go`. Domain-specific validators, prompt fragments,
|
||||||
|
schemas, reference helpers, and codecs also live within that domain tree.
|
||||||
|
|
||||||
|
Each domain exposes one production registration entry point from a sibling
|
||||||
|
registrar package, for example `internal/modules/dnd/register`. The registrar
|
||||||
|
may import the domain root and its child implementations; the domain root does
|
||||||
|
not import its registrar or child packages. This keeps shared types available
|
||||||
|
as `dnd.SpellList` without creating a Go import cycle.
|
||||||
|
|
||||||
|
The `generic` tree is a peer extension family for reusable implementations that
|
||||||
|
contain no concrete source-format or artifact-domain knowledge. Source-format
|
||||||
|
and output-format families, such as Seriatim and JSON output, follow the same
|
||||||
|
domain-first organization even when they do not define a type in the
|
||||||
|
[domain artifact zone](0003-typed-interfaces-with-two-zone-data-model.md#domain-artifact-zone).
|
||||||
|
|
||||||
|
Concrete domain implementation packages do not import another concrete domain.
|
||||||
|
Generic extension packages never import concrete domains. A domain registrar
|
||||||
|
may import domain-neutral generic extension packages to instantiate a reusable
|
||||||
|
strategy for that domain's artifact type; the generic implementation remains
|
||||||
|
unaware of the concrete type's domain semantics. Reuse needed directly by a
|
||||||
|
domain implementation lives in a domain-neutral framework or helper package,
|
||||||
|
not in a peer extension package.
|
||||||
|
|
||||||
|
The application composition root may import multiple registrar packages, and
|
||||||
|
black-box integration tests may compose multiple domains. Other cross-domain
|
||||||
|
reuse occurs through engine contracts and composition-time registration rather
|
||||||
|
than concrete peer-domain imports.
|
||||||
|
|
||||||
|
A domain registrar owns registration of that domain's modules, validators,
|
||||||
|
default validator chains, artifact codecs, schemas, and prompt assets. It does
|
||||||
|
not take ownership of application execution or process behavior.
|
||||||
|
|
||||||
|
## Alternatives considered
|
||||||
|
|
||||||
|
- Group modules by stage. This keeps interchangeable strategies side by side,
|
||||||
|
but scatters a domain's shared artifact model and collaborating extensions
|
||||||
|
across the repository. It is preferable when generic strategy libraries
|
||||||
|
dominate or when the project is primarily a stage-extension framework rather
|
||||||
|
than an application composed from domain suites.
|
||||||
|
- Put both shared types and `Register` in the domain root. This gives the
|
||||||
|
shortest import path but creates an import cycle once child implementations
|
||||||
|
import the root artifact types.
|
||||||
|
|
||||||
|
## Consequences
|
||||||
|
|
||||||
|
The repository layout makes supported domains immediately visible, and adding
|
||||||
|
or extracting a domain affects one cohesive subtree. The CLI composition root
|
||||||
|
depends on a small set of domain registrars instead of every leaf package.
|
||||||
|
|
||||||
|
Package moves must preserve user-visible module and validator keys unless a
|
||||||
|
separate compatibility decision changes them. Shared behavior that cannot be
|
||||||
|
expressed through framework contracts may need to move into a domain-neutral
|
||||||
|
framework package rather than creating a concrete peer-domain import. Registrar
|
||||||
|
packages become explicit composition points for instantiating generic typed
|
||||||
|
strategies, in addition to registering domain-owned implementations.
|
||||||
141
docs/adr/0005-cache-canonical-chunk-plans-by-source.md
Normal file
141
docs/adr/0005-cache-canonical-chunk-plans-by-source.md
Normal file
@@ -0,0 +1,141 @@
|
|||||||
|
# ADR-0005: Cache one canonical chunk plan per source
|
||||||
|
|
||||||
|
**Status:** Accepted
|
||||||
|
**Date:** 2026-07-17
|
||||||
|
|
||||||
|
## Context
|
||||||
|
|
||||||
|
Notarius may run several extraction passes over the same source. A D&D
|
||||||
|
transcript, for example, may first produce NPC artifacts and later produce
|
||||||
|
spell or combat artifacts, with output from an earlier pass supplied as a
|
||||||
|
reference to a later pass.
|
||||||
|
|
||||||
|
An LLM-backed chunker may process an entire, potentially large source in one
|
||||||
|
expensive request. Recomputing boundaries for every pipeline or pass repeats
|
||||||
|
that cost and can make otherwise comparable extraction runs use different
|
||||||
|
source partitions. Stable chunk material also gives later extraction requests
|
||||||
|
a better opportunity to benefit from provider-side prompt caching.
|
||||||
|
|
||||||
|
Chunk boundaries can affect extraction quality. Evidence may span a boundary,
|
||||||
|
overlap may produce duplicates, and different partitions may change the context
|
||||||
|
available to a model. Merge and normalization should remove structural signs
|
||||||
|
of chunking from durable output, but they cannot guarantee recovery of evidence
|
||||||
|
that an extractor did not receive.
|
||||||
|
|
||||||
|
Notarius therefore needs an explicit policy for choosing between automatically
|
||||||
|
applying the latest chunking configuration and preserving one stable partition
|
||||||
|
for repeated work on the same source.
|
||||||
|
|
||||||
|
## Decision
|
||||||
|
|
||||||
|
Notarius assigns one active canonical chunk plan to a source and reuses that
|
||||||
|
plan by default across pipelines and invocations.
|
||||||
|
|
||||||
|
The canonical source identity is derived from the validated generic source
|
||||||
|
document and covers the source-unit identity, order, and content needed to
|
||||||
|
interpret plan boundaries. Input-adapter and chunk-producer identities are
|
||||||
|
recorded as provenance, but the active-plan lookup does not vary with:
|
||||||
|
|
||||||
|
- pipeline identity or selected artifact lanes;
|
||||||
|
- the configured chunk module or its options;
|
||||||
|
- references;
|
||||||
|
- LLM provider, model, profile, prompt, or response schema; or
|
||||||
|
- configuration for later pipeline stages.
|
||||||
|
|
||||||
|
When an active plan exists, Notarius uses it even if the current pipeline
|
||||||
|
configures a different chunk module or different chunk-module settings. The
|
||||||
|
configured chunk module generates a plan only when none exists or when the
|
||||||
|
operator explicitly requests recomputation.
|
||||||
|
|
||||||
|
The framework-owned minimum plan contract is an ordered, non-empty set of
|
||||||
|
source-unit ranges. Each range identifies the inclusive start and end unit for
|
||||||
|
one chunk. A chunk module may also provide namespaced, domain-specific
|
||||||
|
annotations at plan or range scope. Those annotations are stored with the plan
|
||||||
|
and passed through the pipeline when present, but they remain optional.
|
||||||
|
Downstream stages must not assume that annotations associated with the
|
||||||
|
currently configured chunk module are present on a reused plan produced by a
|
||||||
|
different module.
|
||||||
|
|
||||||
|
The cache stores the plan rather than fully materialized chunks. The framework
|
||||||
|
validates a reused plan against the current source and deterministically
|
||||||
|
materializes its ranges into chunks. The same source and plan must produce
|
||||||
|
byte-stable chunk input for later stages.
|
||||||
|
|
||||||
|
Canonical plan storage is a distinct cache surface with an independently
|
||||||
|
configurable location. It is not coupled to the roots or lifecycles of
|
||||||
|
invocation checkpoints, diagnostics, debug artifacts, or durable output. This
|
||||||
|
allows per-user and system-service deployments to apply cache-specific
|
||||||
|
ownership, permissions, placement, and cleanup policy without relocating other
|
||||||
|
Notarius state.
|
||||||
|
|
||||||
|
One mutable active plan is stored under the canonical source identity and
|
||||||
|
retains provenance for the module and relevant runtime inputs that produced it.
|
||||||
|
Refreshing the active plan atomically replaces that one mutable record; readers
|
||||||
|
must observe either the previous complete plan or the replacement complete
|
||||||
|
plan, never a partial update.
|
||||||
|
The effective plan producer is reported separately from the chunk module
|
||||||
|
requested by the current pipeline; reuse must not attribute cached boundaries
|
||||||
|
or annotations to a module that did not produce them.
|
||||||
|
|
||||||
|
Reuse is enabled by default. Operators can explicitly:
|
||||||
|
|
||||||
|
- bypass cached plans for an invocation without changing the active plan; or
|
||||||
|
- recompute a plan with the configured chunk module and make it active for
|
||||||
|
later work.
|
||||||
|
|
||||||
|
Exact storage layout, configuration fields, CLI syntax, publication mechanics,
|
||||||
|
recovery behavior, and diagnostics are implementation and operational
|
||||||
|
contracts rather than part of this decision.
|
||||||
|
|
||||||
|
## Alternatives considered
|
||||||
|
|
||||||
|
- Recompute chunks on every invocation. This always applies the current
|
||||||
|
chunking configuration, but repeats the most expensive stage and weakens
|
||||||
|
provider-side caching and cross-pass comparability.
|
||||||
|
- Cache every distinct chunking request by including module options,
|
||||||
|
references, prompts, profiles, and other runtime inputs in its identity. This
|
||||||
|
closely associates a cached result with its producing request, but reduces
|
||||||
|
reuse and permits boundary drift across operationally different passes.
|
||||||
|
- Key plans by source plus chunk module and options. This shares plans across
|
||||||
|
pipelines using the same strategy, but changing the configured strategy
|
||||||
|
silently selects a different partition rather than preserving one canonical
|
||||||
|
partition for the source.
|
||||||
|
- Require operators to name or supply a plan for every run. Explicit selection
|
||||||
|
is reproducible and may be useful as an advanced operation, but adds friction
|
||||||
|
to the default workflow and does not provide automatic reuse.
|
||||||
|
- Store fully materialized chunks. This simplifies loading, but duplicates
|
||||||
|
source content and couples durable state to the current chunk representation
|
||||||
|
rather than the stable boundary decision.
|
||||||
|
- Store canonical plans beneath the general workspace root. This would reuse an
|
||||||
|
existing location setting, but it couples a reusable application cache to
|
||||||
|
checkpoint, diagnostic, and debug state that have different ownership,
|
||||||
|
sensitivity, retention, and deployment requirements.
|
||||||
|
|
||||||
|
## Consequences
|
||||||
|
|
||||||
|
Independent pipelines and passes over the same source use stable boundaries by
|
||||||
|
default. This reduces repeated LLM work, improves cross-pass comparability, and
|
||||||
|
increases the opportunity for cached provider reads.
|
||||||
|
|
||||||
|
The configured chunk module may not execute, and its settings may have no
|
||||||
|
effect, when an active plan already exists. Domain-specific annotations reflect
|
||||||
|
the plan's original producer and may be absent or differ from those the current
|
||||||
|
module would produce. User-visible provenance must make the effective plan
|
||||||
|
clear.
|
||||||
|
|
||||||
|
A poor or outdated partition remains active until an operator replaces it.
|
||||||
|
This can preserve suboptimal context boundaries and affect extraction recall or
|
||||||
|
duplication even when merge and normalization hide the partition structure in
|
||||||
|
durable output. Stable reuse is an intentional priority over automatically
|
||||||
|
incorporating later chunk-strategy changes.
|
||||||
|
|
||||||
|
The framework gains a durable minimal chunk-plan contract and deterministic
|
||||||
|
materialization responsibility. Chunk modules must separate required boundary
|
||||||
|
output from optional annotations, and downstream modules may rely only on the
|
||||||
|
minimal boundary contract unless a future decision introduces an explicit plan
|
||||||
|
compatibility mechanism.
|
||||||
|
|
||||||
|
Operators must configure and secure canonical plan storage independently from
|
||||||
|
other workspace state when the per-user default is not appropriate. Removing
|
||||||
|
that cache remains recoverable because Notarius can regenerate it from the
|
||||||
|
source, but doing so may repeat an expensive LLM operation.
|
||||||
114
docs/adr/0006-separate-output-cache-and-debug-state.md
Normal file
114
docs/adr/0006-separate-output-cache-and-debug-state.md
Normal file
@@ -0,0 +1,114 @@
|
|||||||
|
# ADR-0006: Separate output, cache, and debug state
|
||||||
|
|
||||||
|
**Status:** Superseded by [ADR-0007](0007-separate-checkpoint-recording-from-reuse.md)
|
||||||
|
**Date:** 2026-07-17
|
||||||
|
|
||||||
|
## Context
|
||||||
|
|
||||||
|
Notarius currently exposes a workspace as a shared parent for checkpoints,
|
||||||
|
debug artifacts, and preferred diagnostics settings. Diagnostics are a second
|
||||||
|
inspection surface with their own enablement, directory, retention, and legacy
|
||||||
|
configuration. Durable output uses a separate CLI-selected root, while the
|
||||||
|
canonical chunk-plan cache introduced by ADR-0005 correctly uses an independent
|
||||||
|
cache root.
|
||||||
|
|
||||||
|
These concepts reflect implementation history more than operator intent. A user
|
||||||
|
must understand differences among workspace state, diagnostics, debug artifacts,
|
||||||
|
checkpoints, and chunk plans before deciding where Notarius may write. Some of
|
||||||
|
those distinctions are important internally: a redacted run summary has a
|
||||||
|
different sensitivity from a trace containing source material, prompts, and
|
||||||
|
model responses. They do not require separate public filesystem categories.
|
||||||
|
|
||||||
|
Notarius needs a smaller state model that communicates why data exists, how it
|
||||||
|
may be treated, and whether it is reconstructible.
|
||||||
|
|
||||||
|
## Decision
|
||||||
|
|
||||||
|
Notarius exposes three filesystem surfaces: output, cache, and debug. The
|
||||||
|
public workspace concept and diagnostics as a separate output surface are
|
||||||
|
removed.
|
||||||
|
|
||||||
|
### Output
|
||||||
|
|
||||||
|
Output is the durable result of a run and the only surface intended for normal
|
||||||
|
consumption. It contains the logical files produced by the output stage,
|
||||||
|
including the maintained result, manifest, warning, and rejection contracts.
|
||||||
|
Output is not cache or inspection state.
|
||||||
|
|
||||||
|
### Cache
|
||||||
|
|
||||||
|
Cache contains reconstructible state used to avoid repeated work or resume an
|
||||||
|
interrupted workflow. Canonical chunk plans and invocation checkpoints are
|
||||||
|
distinct cache families with independent identities, compatibility rules,
|
||||||
|
enablement policies, locations, and cleanup lifecycles.
|
||||||
|
|
||||||
|
ADR-0005 continues to govern canonical chunk-plan selection and reuse. Grouping
|
||||||
|
chunk plans and checkpoints under the public cache category does not permit a
|
||||||
|
checkpoint to compete with canonical plan reuse or couple their storage roots.
|
||||||
|
|
||||||
|
Checkpointing is an invocation policy rather than a prerequisite hidden in
|
||||||
|
persistent workspace configuration. An explicit resume invocation may read
|
||||||
|
compatible checkpoints and record replacement checkpoint state for work it
|
||||||
|
executes. Runs that do not request resume perform no checkpoint I/O.
|
||||||
|
|
||||||
|
### Debug
|
||||||
|
|
||||||
|
Debug is an explicitly requested per-run inspection bundle intended for
|
||||||
|
developers and troubleshooting. It is off by default. When enabled, one bundle
|
||||||
|
contains both redacted run summaries and detailed stage and LLM traces. The
|
||||||
|
internal distinction between a safe summary and a sensitive trace remains, but
|
||||||
|
there is one public enablement and location model.
|
||||||
|
|
||||||
|
Debug data is never a cache input and has no automatic retention policy.
|
||||||
|
Notarius does not create a debug directory unless debug is requested, and it
|
||||||
|
does not automatically delete a requested bundle. Credentials remain redacted
|
||||||
|
at every level, while the bundle as a whole is treated as potentially sensitive
|
||||||
|
because traces may contain source, reference, prompt, model-response, and
|
||||||
|
intermediate artifact content.
|
||||||
|
|
||||||
|
Concise progress, warnings, and failures continue to use stdout or stderr. A
|
||||||
|
run without debug may fail without producing a filesystem inspection record.
|
||||||
|
|
||||||
|
Exact configuration fields, CLI flags, default paths, layouts, compatibility
|
||||||
|
handling, and migration mechanics are configuration and operational contracts
|
||||||
|
rather than part of this decision.
|
||||||
|
|
||||||
|
## Alternatives considered
|
||||||
|
|
||||||
|
- Keep workspace, diagnostics, checkpoints, debug, and chunk-plan cache as
|
||||||
|
separate public concepts. This preserves compatibility and the current safe
|
||||||
|
default-on failure records, but retains overlapping configuration and asks
|
||||||
|
operators to reason about implementation-specific categories.
|
||||||
|
- Keep diagnostics as an always-available redacted operational surface and use
|
||||||
|
debug only for sensitive traces. This distinction is useful for a daemon or
|
||||||
|
managed service with an operational logging contract, but the current CLI can
|
||||||
|
report concise failures on stderr and provide inspection data when explicitly
|
||||||
|
requested.
|
||||||
|
- Put all non-output state beneath one physical root. This minimizes path
|
||||||
|
configuration, but couples reconstructible caches to per-run inspection data
|
||||||
|
and couples cache families whose identity, sensitivity, and cleanup policies
|
||||||
|
differ.
|
||||||
|
- Treat checkpoints as durable run state rather than cache. This emphasizes
|
||||||
|
resumability, but checkpoints are derived, compatibility-checked data that may
|
||||||
|
be deleted and recomputed. Cache more accurately describes their lifecycle.
|
||||||
|
|
||||||
|
## Consequences
|
||||||
|
|
||||||
|
The operator model becomes smaller: normal runs produce output and may use
|
||||||
|
cache; developers explicitly request debug. Public configuration no longer
|
||||||
|
exposes a workspace or overlapping diagnostics and debug systems.
|
||||||
|
|
||||||
|
The implementation retains separate collaborators and serializers where their
|
||||||
|
security or lifecycle boundaries differ. Redacted summaries remain useful as
|
||||||
|
the index to a debug bundle, and chunk plans and checkpoints retain separate
|
||||||
|
stores even though both are cache.
|
||||||
|
|
||||||
|
Existing configuration, environment variables, flags, examples, and
|
||||||
|
documentation require a deliberate compatibility transition. Default-on
|
||||||
|
diagnostic directories disappear, so failures without debug are inspectable
|
||||||
|
only through stderr and any durable output completed before the failure.
|
||||||
|
|
||||||
|
Debug becomes easier to request and substantially more complete, but enabling
|
||||||
|
it creates sensitive files that the operator must protect and remove. Cache
|
||||||
|
cleanup is recoverable but may repeat expensive work, while deleting output is
|
||||||
|
data loss from the user's perspective.
|
||||||
50
docs/adr/0007-separate-checkpoint-recording-from-reuse.md
Normal file
50
docs/adr/0007-separate-checkpoint-recording-from-reuse.md
Normal file
@@ -0,0 +1,50 @@
|
|||||||
|
# ADR-0007: Separate checkpoint recording from reuse
|
||||||
|
|
||||||
|
**Status:** Accepted
|
||||||
|
**Date:** 2026-07-19
|
||||||
|
|
||||||
|
## Context
|
||||||
|
|
||||||
|
ADR-0006 made checkpoint I/O conditional on an explicit `--resume` invocation.
|
||||||
|
That policy requires an operator to anticipate the need for recovery before a
|
||||||
|
run begins. A failed ordinary run cannot reuse completed work because it did not
|
||||||
|
record checkpoints.
|
||||||
|
|
||||||
|
Recording reconstructible state and authorizing reuse are separate operational
|
||||||
|
decisions. Recording consumes storage and retains sensitive derived application
|
||||||
|
data, while reuse may change which module operations execute during a run.
|
||||||
|
|
||||||
|
## Decision
|
||||||
|
|
||||||
|
ADR-0006's separation of output, cache, and debug surfaces remains in effect;
|
||||||
|
this decision supersedes only its checkpoint invocation policy.
|
||||||
|
|
||||||
|
Checkpoint recording is controlled by an explicit persistent Boolean
|
||||||
|
configuration setting and remains disabled by default. When recording is
|
||||||
|
enabled, every run records checkpoint transitions and reusable approved stage
|
||||||
|
results.
|
||||||
|
|
||||||
|
Checkpoint loading remains an invocation policy. Only a run with `--resume`
|
||||||
|
loads and reuses compatible completed work. A recording-enabled run without
|
||||||
|
`--resume` executes every stage normally and never loads checkpoints. A resume
|
||||||
|
request while recording is disabled is rejected.
|
||||||
|
|
||||||
|
The existing checkpoint identities, compatibility rules, payload format,
|
||||||
|
filesystem root behavior, and pipeline collaborator contracts remain unchanged.
|
||||||
|
|
||||||
|
## Alternatives considered
|
||||||
|
|
||||||
|
- Continue coupling reads and writes to `--resume`. This is safe by default but
|
||||||
|
prevents recovery unless resume was anticipated on the earlier run.
|
||||||
|
- Always record checkpoints. This maximizes recovery but creates potentially
|
||||||
|
sensitive state without explicit operator consent.
|
||||||
|
- Add a multi-value recording policy. This preserves the old behavior as an
|
||||||
|
option but adds configuration complexity without a current need.
|
||||||
|
|
||||||
|
## Consequences
|
||||||
|
|
||||||
|
Operators can opt into recovery-ready runs while keeping checkpoint reuse
|
||||||
|
explicit. Enabled successful, rejected, and failed runs may all leave sensitive
|
||||||
|
checkpoint state, so operators remain responsible for access and retention.
|
||||||
|
Disabled configurations perform no checkpoint I/O, and `--resume` requires the
|
||||||
|
operator to enable recording first.
|
||||||
50
docs/adr/0008-ordered-pipeline-steps.md
Normal file
50
docs/adr/0008-ordered-pipeline-steps.md
Normal file
@@ -0,0 +1,50 @@
|
|||||||
|
# ADR-0008: Bounded ordered pipeline steps and explicit artifact references
|
||||||
|
|
||||||
|
**Status:** Accepted
|
||||||
|
**Date:** 2026-07-21
|
||||||
|
|
||||||
|
## Context
|
||||||
|
|
||||||
|
Notarius currently models one pipeline-wide input, chunking plan, artifact
|
||||||
|
lanes, and output boundary. Some workflows need a deterministic handoff from
|
||||||
|
one set of normalized artifacts to a later set of artifacts, such as using
|
||||||
|
extracted NPC records while grounding later combat events. The workflow needs
|
||||||
|
an explicit topology without turning the pipeline into a general-purpose
|
||||||
|
workflow engine.
|
||||||
|
|
||||||
|
## Decision
|
||||||
|
|
||||||
|
Add an ordered collection of pipeline steps. Each step owns one or more
|
||||||
|
artifact lanes, and lanes within a step retain the existing independent
|
||||||
|
execution model. The pipeline continues to have one input, chunk plan, output,
|
||||||
|
and failure boundary. Steps are barriers: a later step may consume only
|
||||||
|
normalized artifacts from an earlier step.
|
||||||
|
|
||||||
|
Generated references use an explicit step-and-lane selector. Reference slots
|
||||||
|
declare the generated artifact kinds and media types they accept. The resolver
|
||||||
|
validates the topology, ordering, lane identity, artifact kind, schema, and
|
||||||
|
codec compatibility before execution. External references remain supported as
|
||||||
|
path sources, and the legacy top-level artifact map is interpreted as an
|
||||||
|
implicit `default` step.
|
||||||
|
|
||||||
|
Pipeline-level references may not select generated artifacts. General DAGs,
|
||||||
|
branches, loops, conditional execution, joins, and inferred dependencies are
|
||||||
|
not part of this model.
|
||||||
|
|
||||||
|
## Alternatives considered
|
||||||
|
|
||||||
|
- A general DAG would provide more flexibility but would also require a new
|
||||||
|
scheduler, lifecycle model, failure semantics, and provenance model.
|
||||||
|
- Separate pipeline runs connected through filesystem paths would lose the
|
||||||
|
static topology and typed compatibility checks.
|
||||||
|
- Inferring dependencies from module or lane names would make ordering and
|
||||||
|
configuration errors difficult to detect reliably.
|
||||||
|
|
||||||
|
## Consequences
|
||||||
|
|
||||||
|
The resolved pipeline has a deterministic, inspectable topology and can
|
||||||
|
include it in its identity digest. Configuration validation can reject invalid
|
||||||
|
generated bindings before any work begins. Existing single-step profiles keep
|
||||||
|
their behavior through the implicit `default` step. Execution handoff and
|
||||||
|
multi-step scheduling require follow-up work in the runner and checkpoint
|
||||||
|
layers.
|
||||||
@@ -0,0 +1,98 @@
|
|||||||
|
# ADR-0009: Prefer minimal evidence-grounded extraction artifacts
|
||||||
|
|
||||||
|
**Status:** Accepted
|
||||||
|
**Date:** 2026-07-22
|
||||||
|
|
||||||
|
## Context
|
||||||
|
|
||||||
|
Notarius is intended to extract structured facts from source material. Several
|
||||||
|
early D&D artifacts grew to include descriptive prose, inferred relationships,
|
||||||
|
immediate outcomes, summaries, and other enrichment alongside the facts that
|
||||||
|
identify an event or entity. Those fields make one model call responsible for
|
||||||
|
both extraction and synthesis.
|
||||||
|
|
||||||
|
In practice, the richer contracts have produced overlapping or weakly grounded
|
||||||
|
fields and have made structurally valid, semantically coherent output harder for
|
||||||
|
cost-effective smaller models. They also increase prompt size, validation and
|
||||||
|
normalization policy, durable schema surface, downstream coupling, and the
|
||||||
|
number of claims whose provenance must be evaluated.
|
||||||
|
|
||||||
|
The application needs a consistent rule for deciding what belongs in an
|
||||||
|
extractor before redesigning the current D&D spell, NPC, and combat-turn
|
||||||
|
contracts or adding new artifact families.
|
||||||
|
|
||||||
|
## Decision
|
||||||
|
|
||||||
|
An extraction module answers one narrowly stated question and returns the
|
||||||
|
smallest durable structured artifact that usefully answers it.
|
||||||
|
|
||||||
|
Every model-produced field in an extraction artifact must:
|
||||||
|
|
||||||
|
- be necessary to answer the extractor's stated question or serve a known
|
||||||
|
downstream consumer;
|
||||||
|
- represent a fact or bounded classification that can be supported directly by
|
||||||
|
cited source ranges;
|
||||||
|
- remain independently meaningful without model-generated explanatory prose;
|
||||||
|
and
|
||||||
|
- justify the additional prompt, schema, validation, normalization, and
|
||||||
|
compatibility surface it creates.
|
||||||
|
|
||||||
|
Source references are required provenance for extracted records. Auxiliary
|
||||||
|
references may disambiguate identities or canonical names, but they do not
|
||||||
|
establish source facts and are not copied into evidence.
|
||||||
|
|
||||||
|
Extraction artifacts do not include narrative summaries, general analysis,
|
||||||
|
speculative enrichment, inferred biography or relationships, or redundant
|
||||||
|
free-text descriptions by default. When such output has a demonstrated use, it
|
||||||
|
belongs in an explicitly named extraction, classification, enrichment, or
|
||||||
|
analysis module with its own contract and evidence policy.
|
||||||
|
|
||||||
|
Occurrence-level facts are not forced into entity-level attributes. A fact
|
||||||
|
that can change between encounters, such as an NPC's role in a scene, belongs
|
||||||
|
on an occurrence artifact rather than as one scalar property of a normalized
|
||||||
|
NPC registry entry.
|
||||||
|
|
||||||
|
Deterministic mapping and normalization may assign application-owned
|
||||||
|
identifiers, canonicalize known catalog values, order and deduplicate evidence,
|
||||||
|
and collapse records under an explicit identity rule. They must not manufacture
|
||||||
|
removed descriptive fields or synthesize missing claims to satisfy an older
|
||||||
|
contract.
|
||||||
|
|
||||||
|
This is a default design rule, not a prohibition on rich artifacts. A richer
|
||||||
|
field is appropriate when its consumer, evidence semantics, and ownership are
|
||||||
|
explicit.
|
||||||
|
|
||||||
|
## Alternatives considered
|
||||||
|
|
||||||
|
- Keep rich schemas and improve prompts or use larger models. This retains
|
||||||
|
potentially convenient prose but does not resolve overlapping field
|
||||||
|
responsibilities, weak provenance, higher cost, or unnecessary downstream
|
||||||
|
coupling.
|
||||||
|
- Make enrichment fields optional. This reduces rejection pressure but leaves
|
||||||
|
ambiguous artifact semantics and inconsistent records, and many strict
|
||||||
|
structured-output providers still require nullable placeholders.
|
||||||
|
- Keep minimal private LLM schemas while preserving rich durable artifacts.
|
||||||
|
Deterministic code would have to invent, default, or separately derive the
|
||||||
|
missing fields, hiding synthesis behind the extraction boundary.
|
||||||
|
- Use one broad session-analysis module. This reduces the number of lanes but
|
||||||
|
couples unrelated facts, schemas, retries, evaluation, and downstream
|
||||||
|
consumers into one model call.
|
||||||
|
|
||||||
|
## Consequences
|
||||||
|
|
||||||
|
Extraction prompts and response schemas become smaller, more focused, and more
|
||||||
|
suitable for lower-cost models. Artifacts carry fewer unsupported claims, and
|
||||||
|
their evidence and validation policies become easier to explain and evaluate.
|
||||||
|
Independent extractors can evolve, retry, and be consumed without requiring
|
||||||
|
unrelated enrichment.
|
||||||
|
|
||||||
|
Some descriptive convenience fields will disappear from primary artifacts.
|
||||||
|
Consumers that genuinely need them may require a separate module and explicit
|
||||||
|
pipeline step. Entity registries may no longer resolve aliases or relationships
|
||||||
|
unless a dedicated, evidence-grounded capability supplies them.
|
||||||
|
|
||||||
|
Removing durable fields is a schema compatibility change. Each affected
|
||||||
|
artifact requires an explicit version and reference policy; private prompt
|
||||||
|
changes alone are insufficient. Current-behavior integration and internal
|
||||||
|
documentation must change with implementation, while the roadmap owns the
|
||||||
|
proposed contract until then.
|
||||||
217
docs/cli.md
217
docs/cli.md
@@ -1,129 +1,158 @@
|
|||||||
# CLI Reference
|
# CLI Reference
|
||||||
|
|
||||||
This is the canonical reference for the implemented Notarius command-line
|
This is the canonical reference for the implemented Notarius command-line
|
||||||
interface.
|
interface. For the shortest successful run, see the [README](../README.md).
|
||||||
|
Configuration fields, discovery rules, and selectable module keys are defined
|
||||||
|
in [Configuration](config.md); runtime state and recovery procedures are
|
||||||
|
defined in [Operations](operations.md).
|
||||||
|
|
||||||
## Quick Run
|
## Command Summary
|
||||||
|
|
||||||
```sh
|
~~~
|
||||||
NOTARIUS_LLM_DEFAULT_BASE_URL=http://127.0.0.1:8080/v1 \
|
|
||||||
NOTARIUS_LLM_DEFAULT_MODEL=your-model \
|
|
||||||
go run ./cmd/notarius run dnd-session \
|
|
||||||
--config examples/dnd-spells.config.yml \
|
|
||||||
--input examples/seriatim-minimal-transcript.json
|
|
||||||
```
|
|
||||||
|
|
||||||
Set `NOTARIUS_LLM_DEFAULT_API_KEY` if the OpenAI-compatible provider requires
|
|
||||||
a bearer token.
|
|
||||||
|
|
||||||
## Commands
|
|
||||||
|
|
||||||
```text
|
|
||||||
notarius help
|
notarius help
|
||||||
notarius run <pipeline-id> --input path/to/source.json [--config path/to/config.yml] [--only lane-a,lane-b]
|
notarius run <pipeline-id> --input path/to/source.json [--json] [flags]
|
||||||
notarius config validate --config path/to/config.yml [--pipeline pipeline-id] [--only lane-a,lane-b]
|
notarius config validate [--config path/to/config.yml] [--pipeline pipeline-id] [--only lane-a,lane-b]
|
||||||
notarius pipelines list --config path/to/config.yml [--json]
|
notarius pipelines list [--config path/to/config.yml] [--json]
|
||||||
```
|
~~~
|
||||||
|
|
||||||
Running `notarius` with no arguments, `notarius help`, `notarius --help`, or
|
Running Notarius without arguments, or with **help**, **--help**, or **-h**,
|
||||||
`notarius -h` prints usage and exits successfully.
|
writes the command summary to standard output and exits with status 0.
|
||||||
|
|
||||||
## `run`
|
## run
|
||||||
|
|
||||||
`notarius run <pipeline-id>` executes a configured pipeline against one input
|
~~~
|
||||||
file.
|
notarius run <pipeline-id> --input path/to/source.json [--json] [flags]
|
||||||
|
~~~
|
||||||
|
|
||||||
Flags:
|
The **run** command executes the named pipeline for one input file. The
|
||||||
|
pipeline ID and **--input** are required.
|
||||||
|
|
||||||
- `--input path`: required source input file.
|
| Flag | Meaning |
|
||||||
- `--config path`: config file path. If omitted, Notarius checks
|
| --- | --- |
|
||||||
`NOTARIUS_CONFIG`, then `/usr/local/etc/notarius/config.yml`.
|
| **--config path** | Use this configuration file. When omitted, configuration discovery applies; see [Configuration](config.md). |
|
||||||
- `--only lane-a,lane-b`: run only the named artifact lanes. Values are
|
| **--input path** | Source input file to process. Required. |
|
||||||
comma-separated and must be non-empty.
|
| **--output-dir path** | Override the configured output root for this run. |
|
||||||
- `--output-dir path`: output root. The run writes to `<path>/<run-id>/`.
|
| **--json** | Write the successful run-result receipt as JSON to standard output. |
|
||||||
Defaults to `./notarius-output`.
|
| **--chunk_cache auto\|bypass\|refresh** | Override chunk-plan cache handling for this run. |
|
||||||
- `--diagnostics-dir path`: diagnostics work directory override for this
|
| **--resume** | Reuse compatible recorded checkpoints when checkpoint recording is enabled. |
|
||||||
invocation.
|
| **--recompute-step step-id** | With **--resume**, recompute the selected ordered step and its dependent lanes. It cannot be combined with **--only**. |
|
||||||
- `--llm-profile id`: override every effective module binding to use one LLM
|
| **--debug** | Retain a debug bundle for this run. |
|
||||||
profile.
|
| **--debug-dir path** | Override the debug-bundle root. Requires **--debug**. |
|
||||||
|
| **--only lane-a,lane-b** | Run only the selected comma-separated artifact lanes when that selection is valid for the configured pipeline. |
|
||||||
|
| **--llm-profile id** | Override effective LLM-capable module bindings with one configured profile. |
|
||||||
|
| **--session-id id** | Supply a non-empty prompt session identifier to LLM-backed module calls. |
|
||||||
|
| **--reference selector=path** | Add or replace a file reference binding. Repeatable. |
|
||||||
|
| **--without-reference selector** | Remove a configured optional reference binding. Repeatable. |
|
||||||
|
|
||||||
On success, the command prints the completed pipeline ID, approved and rejected
|
**--chunk_cache** accepts only **auto**, **bypass**, or **refresh**.
|
||||||
artifact counts, and the output directory. If the run completes with warnings,
|
**--debug-dir**, **--output-dir**, **--session-id**, and
|
||||||
the warning count is printed to stderr.
|
**--recompute-step** reject explicit empty values. **--recompute-step**
|
||||||
|
requires **--resume**; checkpoint requirements and reuse behavior are
|
||||||
|
documented in [Operations](operations.md).
|
||||||
|
|
||||||
For durable output, diagnostics, retention, and failure inspection, see
|
### Reference selectors
|
||||||
[Operations](operations.md).
|
|
||||||
|
|
||||||
The current `run` command requires the resolved pipeline to use exactly one
|
Use **--reference** only for a reference slot declared by the selected
|
||||||
distinct LLM profile after defaults and overrides are applied.
|
configured target. The accepted selector forms are:
|
||||||
|
|
||||||
## `config validate`
|
| Form | Target |
|
||||||
|
| --- | --- |
|
||||||
|
| slot=path | The unique selected target that declares slot. |
|
||||||
|
| chunk.slot=path | The chunker. |
|
||||||
|
| merge.slot=path | The unique selected merger that declares slot. |
|
||||||
|
| lane.slot=path | The unique extractor, merger, or normalizer in lane that declares slot. |
|
||||||
|
| lane.extract.slot=path | The extractor in lane. |
|
||||||
|
| lane.merge.slot=path | The merger in lane. |
|
||||||
|
| lane.normalize.slot=path | The normalizer in lane. |
|
||||||
|
|
||||||
`notarius config validate` loads and validates configuration.
|
**--without-reference** uses the same selector forms without =path. Slot
|
||||||
|
names, requiredness, and configured bindings are part of the
|
||||||
|
[configuration contract](config.md).
|
||||||
|
|
||||||
Flags:
|
### Run output
|
||||||
|
|
||||||
- `--config path`: config file path. If omitted, discovery uses
|
Without **--json**, standard output contains the completed pipeline ID, counts
|
||||||
`NOTARIUS_CONFIG`, then `/usr/local/etc/notarius/config.yml`.
|
of normalized and rejected outputs, and the output directory. A debug-enabled
|
||||||
- `--pipeline pipeline-id`: additionally resolve one configured pipeline against
|
run also prints its debug-bundle path to standard output. A successful run with
|
||||||
the production module catalog.
|
warnings reports the warning count to standard error. The published JSON bundle
|
||||||
- `--only lane-a,lane-b`: validate resolution for selected artifact lanes. This
|
is defined by the [JSON output contract](integrations/json-output.md).
|
||||||
flag requires `--pipeline`.
|
|
||||||
|
With **--json**, successful standard output is exactly one
|
||||||
|
`notarius.run-result.v1` JSON document followed by a newline, with no
|
||||||
|
human-oriented status or debug-path line. Its fields and compatibility policy
|
||||||
|
are defined by the [run-result contract](integrations/run-result.md). A caller
|
||||||
|
must check for exit status 0 before decoding this output; a failed write can
|
||||||
|
leave incomplete standard-output bytes that are not a result document.
|
||||||
|
|
||||||
|
Example:
|
||||||
|
|
||||||
|
~~~
|
||||||
|
OPENROUTER_API_KEY=your-api-key \
|
||||||
|
go run ./cmd/notarius run dnd-session \
|
||||||
|
--config examples/dnd-minimal.config.yml \
|
||||||
|
--input examples/seriatim-minimal-transcript.json
|
||||||
|
~~~
|
||||||
|
|
||||||
|
## config validate
|
||||||
|
|
||||||
|
~~~
|
||||||
|
notarius config validate [--config path/to/config.yml] [--pipeline pipeline-id] [--only lane-a,lane-b]
|
||||||
|
~~~
|
||||||
|
|
||||||
|
This command loads and validates a configuration. With **--pipeline**, it also
|
||||||
|
resolves that pipeline against the production module catalog. **--only** selects
|
||||||
|
lanes during that resolution and requires **--pipeline**.
|
||||||
|
|
||||||
|
Success is written to standard output as either config "<path>" is valid or
|
||||||
|
config "<path>" is valid for pipeline "<pipeline-id>".
|
||||||
|
|
||||||
Examples:
|
Examples:
|
||||||
|
|
||||||
```sh
|
~~~
|
||||||
go run ./cmd/notarius config validate \
|
go run ./cmd/notarius config validate \
|
||||||
--config examples/dnd-spells.config.yml
|
--config examples/dnd-minimal.config.yml \
|
||||||
|
--pipeline dnd-session
|
||||||
|
|
||||||
|
OPENROUTER_API_KEY=validation-placeholder \
|
||||||
go run ./cmd/notarius config validate \
|
go run ./cmd/notarius config validate \
|
||||||
--config examples/dnd-spells.config.yml \
|
--config examples/dnd-complete.config.yml \
|
||||||
--pipeline dnd-session \
|
--pipeline dnd-session
|
||||||
--only spells
|
~~~
|
||||||
```
|
|
||||||
|
|
||||||
## `pipelines list`
|
The placeholder in the second command is sufficient only for offline
|
||||||
|
validation; it cannot run a provider-backed pipeline.
|
||||||
|
|
||||||
`notarius pipelines list` prints configured pipeline IDs in sorted order.
|
## pipelines list
|
||||||
|
|
||||||
Flags:
|
~~~
|
||||||
|
notarius pipelines list [--config path/to/config.yml] [--json]
|
||||||
|
~~~
|
||||||
|
|
||||||
- `--config path`: config file path. If omitted, discovery uses
|
This command lists configured pipeline IDs in sorted order. By default, it
|
||||||
`NOTARIUS_CONFIG`, then `/usr/local/etc/notarius/config.yml`.
|
writes one ID per line to standard output. **--json** writes an object shaped as
|
||||||
- `--json`: print `{"pipelines":[...]}` instead of one ID per line.
|
{"pipelines":[...]} instead.
|
||||||
|
|
||||||
Examples:
|
~~~
|
||||||
|
|
||||||
```sh
|
|
||||||
go run ./cmd/notarius pipelines list \
|
go run ./cmd/notarius pipelines list \
|
||||||
--config examples/dnd-spells.config.yml
|
--config examples/dnd-minimal.config.yml
|
||||||
|
~~~
|
||||||
|
|
||||||
go run ./cmd/notarius pipelines list \
|
## Output Streams And Exit Statuses
|
||||||
--config examples/dnd-spells.config.yml \
|
|
||||||
--json
|
|
||||||
```
|
|
||||||
|
|
||||||
## Exit Codes
|
Successful commands write their primary result to standard output. Warnings and
|
||||||
|
errors are written to standard error.
|
||||||
|
|
||||||
- `0`: command succeeded.
|
For **run --json**, warnings remain on standard error and standard output is a
|
||||||
- `1`: command syntax was valid, but loading config, resolving modules, running
|
machine-readable success result only. Syntax and runtime diagnostics remain on
|
||||||
the pipeline, calling the provider, writing output, or writing diagnostics
|
standard error. Parse the result only after the process exits with status 0.
|
||||||
failed.
|
|
||||||
- `2`: command syntax was invalid, a command was unknown, a required argument
|
|
||||||
was missing, or a flag value was malformed.
|
|
||||||
|
|
||||||
## Implemented Production Pipeline Modules
|
| Status | Meaning |
|
||||||
|
| --- | --- |
|
||||||
|
| 0 | The command completed successfully, including root help. |
|
||||||
|
| 1 | Command syntax was valid but configuration loading or validation, pipeline resolution or execution, provider use, output, or requested debug handling failed. |
|
||||||
|
| 2 | The command or flag syntax was invalid, including unknown commands, missing required arguments, invalid flag values, or invalid flag combinations. |
|
||||||
|
|
||||||
The production CLI currently registers these module keys:
|
The root help spellings are the supported help path. Invoking **--help** on
|
||||||
|
**run**, **config validate**, or **pipelines list** is handled by the flag
|
||||||
- input: `seriatim`
|
parser as a usage error: it writes an error to standard error and exits with
|
||||||
- chunk: `generic`
|
status 2.
|
||||||
- extract: `dnd/spells`
|
|
||||||
- merge: `appendorder`
|
|
||||||
- normalize: `noop`
|
|
||||||
- output: `json`
|
|
||||||
|
|
||||||
The production CLI does not currently register validator modules.
|
|
||||||
|
|
||||||
For YAML structure, defaults, environment overrides, and module binding syntax,
|
|
||||||
see [Configuration](config.md).
|
|
||||||
|
|||||||
498
docs/config.md
498
docs/config.md
@@ -1,213 +1,371 @@
|
|||||||
# Configuration
|
# Configuration
|
||||||
|
|
||||||
This is the canonical reference for implemented Notarius configuration.
|
This is the canonical reference for Notarius configuration. Configuration files
|
||||||
|
are YAML and must declare version 3. They select pipelines and their modules;
|
||||||
|
the [CLI reference](cli.md) owns invocation syntax, and
|
||||||
|
[Operations](operations.md) owns run-state procedures.
|
||||||
|
|
||||||
Notarius reads YAML config files with `version: 1`. File config is applied over
|
## Configuration Discovery And Precedence
|
||||||
built-in defaults, then environment overrides are applied.
|
|
||||||
|
|
||||||
## Discovery
|
Commands that load configuration choose a file in this order:
|
||||||
|
|
||||||
Commands that accept `--config` load configuration in this order:
|
1. a non-empty **--config** CLI value;
|
||||||
|
2. a non-empty **NOTARIUS_CONFIG** environment value;
|
||||||
|
3. the installed default file at **/usr/local/etc/notarius/config.yml**, when
|
||||||
|
it exists.
|
||||||
|
|
||||||
1. the `--config` path, when provided;
|
The command fails if none of these paths provides a configuration file.
|
||||||
2. `NOTARIUS_CONFIG`, when set to a non-empty path;
|
|
||||||
3. `/usr/local/etc/notarius/config.yml`.
|
|
||||||
|
|
||||||
If none is available, the command fails with a config file not found error.
|
For configuration values, precedence is:
|
||||||
|
|
||||||
## Minimal Example
|
1. built-in defaults;
|
||||||
|
2. the selected YAML file;
|
||||||
|
3. supported operational environment variables; and
|
||||||
|
4. the CLI run overrides that apply to a command.
|
||||||
|
|
||||||
```yaml
|
Environment variables do not provide a second configuration schema. They only
|
||||||
version: 1
|
override the fields listed below.
|
||||||
llm_profiles:
|
|
||||||
default:
|
|
||||||
provider: openai-compatible
|
|
||||||
base_url: http://127.0.0.1:8080/v1
|
|
||||||
model: your-model
|
|
||||||
pipelines:
|
|
||||||
dnd-session:
|
|
||||||
input: seriatim
|
|
||||||
chunk:
|
|
||||||
module: generic
|
|
||||||
options:
|
|
||||||
max_units: 50
|
|
||||||
artifacts:
|
|
||||||
spells:
|
|
||||||
extract: dnd/spells
|
|
||||||
```
|
|
||||||
|
|
||||||
The maintained fixture is [examples/dnd-spells.config.yml](../examples/dnd-spells.config.yml).
|
## Maintained Examples
|
||||||
|
|
||||||
## Top-Level Fields
|
- [Minimal D&D configuration](../examples/dnd-minimal.config.yml) is a
|
||||||
|
single-lane Seriatim-to-spell pipeline.
|
||||||
|
- [Complete D&D configuration](../examples/dnd-complete.config.yml) uses
|
||||||
|
ordered steps, all implemented D&D lanes, generated references, state
|
||||||
|
settings, and bounded LLM concurrency.
|
||||||
|
|
||||||
- `version`: required. The only supported value is `1`.
|
Use these complete files as starting points rather than combining the
|
||||||
- `llm_profiles`: optional map of LLM profile IDs to profile settings.
|
illustrative fragments in this reference.
|
||||||
- `pipelines`: optional map of pipeline IDs to pipeline definitions.
|
|
||||||
- `concurrency`: optional global concurrency settings.
|
|
||||||
- `diagnostics`: optional diagnostics settings.
|
|
||||||
|
|
||||||
Unknown YAML fields are rejected.
|
## File Shape And Defaults
|
||||||
|
|
||||||
## Defaults
|
Unknown fields, duplicate mapping keys, empty identifiers, and identifiers that
|
||||||
|
become duplicates after trimming whitespace are rejected. Every top-level field
|
||||||
|
other than **version** is optional.
|
||||||
|
|
||||||
Built-in defaults:
|
| Field | Type | Default | Rules |
|
||||||
|
| --- | --- | --- | --- |
|
||||||
|
| **version** | integer | none | Required; must be 3. |
|
||||||
|
| **scriptorium** | object | none | Profile source configuration. |
|
||||||
|
| **pipelines** | map | empty | Maps pipeline IDs to pipeline definitions. |
|
||||||
|
| **concurrency** | object | see below | Global LLM and extraction limits. |
|
||||||
|
| **output** | object | see below | Published output settings. |
|
||||||
|
| **cache** | object | see below | Chunk-plan and checkpoint settings. |
|
||||||
|
| **debug** | object | see below | Debug-bundle root only; it does not enable capture. |
|
||||||
|
|
||||||
```yaml
|
Built-in defaults are:
|
||||||
llm_profiles:
|
|
||||||
default:
|
| Field | Default |
|
||||||
provider: openai-compatible
|
| --- | --- |
|
||||||
timeout: 600
|
| **concurrency.total_llm** | 1 |
|
||||||
max_retries: 3
|
| **concurrency.stage_workers.extract** | Effective **total_llm** |
|
||||||
max_concurrency: 1
|
| **output.directory** | **./notarius-output** |
|
||||||
|
| **cache.chunk_plans.mode** | **auto** |
|
||||||
|
| **cache.chunk_plans.directory** | Empty, selecting the per-user chunk-plan root |
|
||||||
|
| **cache.checkpoints.enabled** | false |
|
||||||
|
| **cache.checkpoints.directory** | Empty, selecting the per-user checkpoint root |
|
||||||
|
| **debug.directory** | **./notarius-debug** |
|
||||||
|
|
||||||
|
An empty cache directory in YAML deliberately selects the corresponding
|
||||||
|
per-user root. An explicit empty output or debug directory is invalid.
|
||||||
|
|
||||||
|
## Scriptorium Profiles
|
||||||
|
|
||||||
|
The optional **scriptorium** object selects one source of profile definitions:
|
||||||
|
|
||||||
|
| Field | Type | Rules |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| **profile_dir** | string | Non-empty directory containing profile files. |
|
||||||
|
| **profile_file** | string | Non-empty profile file. |
|
||||||
|
|
||||||
|
Set at most one of these fields. Profile IDs used by a binding must be available
|
||||||
|
from the selected Scriptorium profile source when the pipeline is resolved.
|
||||||
|
Keep credentials out of this file: configure a profile to read its credential
|
||||||
|
from an environment variable, then set that environment variable only in the
|
||||||
|
run environment.
|
||||||
|
|
||||||
|
## Operational Environment Variables
|
||||||
|
|
||||||
|
These variables are applied after YAML values:
|
||||||
|
|
||||||
|
| Variable | Overrides | Rules |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| **NOTARIUS_TOTAL_LLM_CONCURRENCY** | **concurrency.total_llm** | Integer. |
|
||||||
|
| **NOTARIUS_STAGE_WORKERS_EXTRACT** | **concurrency.stage_workers.extract** | Integer. |
|
||||||
|
| **NOTARIUS_OUTPUT_DIR** | **output.directory** | Non-empty path. |
|
||||||
|
| **NOTARIUS_CACHE_CHUNK_PLANS_MODE** | **cache.chunk_plans.mode** | **auto**, **bypass**, or **refresh**. |
|
||||||
|
| **NOTARIUS_CACHE_CHUNK_PLANS_DIR** | **cache.chunk_plans.directory** | Non-empty path. |
|
||||||
|
| **NOTARIUS_CACHE_CHECKPOINTS_DIR** | **cache.checkpoints.directory** | Non-empty path. |
|
||||||
|
| **NOTARIUS_DEBUG_DIR** | **debug.directory** | Non-empty path. |
|
||||||
|
|
||||||
|
Integer values are trimmed then parsed as base-10 integers. Directory and
|
||||||
|
output values reject NUL characters. **NOTARIUS_CONFIG** participates only in
|
||||||
|
configuration discovery.
|
||||||
|
|
||||||
|
## Concurrency, Output, Cache, And Debug
|
||||||
|
|
||||||
|
~~~yaml
|
||||||
concurrency:
|
concurrency:
|
||||||
total_llm: 1
|
total_llm: 2
|
||||||
diagnostics:
|
stage_workers:
|
||||||
work_dir: /tmp/notarius
|
extract: 2
|
||||||
retention: auto
|
output:
|
||||||
```
|
directory: ./notarius-output
|
||||||
|
cache:
|
||||||
|
chunk_plans:
|
||||||
|
mode: auto
|
||||||
|
directory: ./notarius-cache/chunk-plans
|
||||||
|
checkpoints:
|
||||||
|
enabled: true
|
||||||
|
directory: ./notarius-cache/checkpoints
|
||||||
|
debug:
|
||||||
|
directory: ./notarius-debug
|
||||||
|
~~~
|
||||||
|
|
||||||
No pipelines are built in. A run requires a configured pipeline.
|
**concurrency.total_llm** must be greater than zero. The only supported
|
||||||
|
**concurrency.stage_workers** key is **extract**; its value must be from 1
|
||||||
|
through **total_llm**. When omitted, it is recalculated from the effective
|
||||||
|
**total_llm** after YAML and environment precedence.
|
||||||
|
|
||||||
## LLM Profiles
|
**cache.chunk_plans.mode** accepts **auto**, **bypass**, or **refresh**.
|
||||||
|
**cache.checkpoints.enabled** is a boolean. The CLI can override the output
|
||||||
Each `llm_profiles` entry may contain:
|
directory and chunk-plan mode for one run; see [CLI reference](cli.md#run).
|
||||||
|
|
||||||
- `provider`: optional provider key. Empty means `openai-compatible`; any other
|
|
||||||
non-empty value must be `openai-compatible`.
|
|
||||||
- `base_url`: provider base URL. Required for actual LLM calls.
|
|
||||||
- `model`: provider model name. Required for actual LLM calls.
|
|
||||||
- `api_key_env`: environment variable name to read for the API key.
|
|
||||||
- `timeout`: request timeout as whole seconds or a Go-style duration string such
|
|
||||||
as `10m`.
|
|
||||||
- `max_retries`: retry count for provider calls. Must be zero or greater.
|
|
||||||
- `max_concurrency`: per-profile LLM concurrency. Must be zero or greater; when
|
|
||||||
zero, Notarius uses `concurrency.total_llm`.
|
|
||||||
|
|
||||||
Raw API keys are not accepted as file config fields. Use `api_key_env` or an
|
|
||||||
environment override.
|
|
||||||
|
|
||||||
## Environment Overrides
|
|
||||||
|
|
||||||
These environment variables are applied after the config file:
|
|
||||||
|
|
||||||
- `NOTARIUS_CONFIG`: config discovery path.
|
|
||||||
- `NOTARIUS_LLM_DEFAULT_API_KEY`: API key for the `default` LLM profile.
|
|
||||||
- `NOTARIUS_LLM_DEFAULT_BASE_URL`: base URL for the `default` LLM profile.
|
|
||||||
- `NOTARIUS_LLM_DEFAULT_MODEL`: model for the `default` LLM profile.
|
|
||||||
- `NOTARIUS_LLM_DEFAULT_TIMEOUT_SECONDS`: integer timeout seconds for the
|
|
||||||
`default` LLM profile.
|
|
||||||
- `NOTARIUS_LLM_DEFAULT_MAX_RETRIES`: integer retry count for the `default` LLM
|
|
||||||
profile.
|
|
||||||
- `NOTARIUS_LLM_DEFAULT_MAX_CONCURRENCY`: integer max concurrency for the
|
|
||||||
`default` LLM profile.
|
|
||||||
- `NOTARIUS_TOTAL_LLM_CONCURRENCY`: integer global LLM concurrency.
|
|
||||||
- `NOTARIUS_WORK_DIR`: diagnostics work directory.
|
|
||||||
- `NOTARIUS_DIAGNOSTICS_RETENTION`: diagnostics retention mode.
|
|
||||||
|
|
||||||
Integer environment values must parse as base-10 integers.
|
|
||||||
|
|
||||||
## Pipelines
|
## Pipelines
|
||||||
|
|
||||||
A pipeline defines the fixed Notarius workflow:
|
Each **pipelines** entry has a unique, non-empty ID and the following shape:
|
||||||
|
|
||||||
```text
|
~~~yaml
|
||||||
input -> chunk -> extract -> merge -> normalize -> output
|
pipelines:
|
||||||
```
|
dnd-session:
|
||||||
|
|
||||||
Pipeline fields:
|
|
||||||
|
|
||||||
- `input`: required module binding.
|
|
||||||
- `chunk`: optional module binding. Default module is `generic`.
|
|
||||||
- `artifacts`: required for pipeline resolution. It maps artifact lane IDs to
|
|
||||||
lane definitions.
|
|
||||||
- `output`: optional module binding. Default module is `json`.
|
|
||||||
|
|
||||||
Artifact lane fields:
|
|
||||||
|
|
||||||
- `extract`: required module binding.
|
|
||||||
- `merge`: optional module binding. Default module is `appendorder`.
|
|
||||||
- `normalize`: optional module binding. Default module is `noop`.
|
|
||||||
- `validators`: optional list of module bindings. The production CLI currently
|
|
||||||
does not register validator modules.
|
|
||||||
|
|
||||||
`notarius run` and `notarius config validate --pipeline` resolve the pipeline
|
|
||||||
against the production module catalog and fail fast for unknown or incompatible
|
|
||||||
module keys.
|
|
||||||
|
|
||||||
## Module Bindings
|
|
||||||
|
|
||||||
Every module binding may use shorthand:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
input: seriatim
|
input: seriatim
|
||||||
```
|
chunk: generic
|
||||||
|
output: json
|
||||||
|
artifacts:
|
||||||
|
spells:
|
||||||
|
extract: dnd/spells
|
||||||
|
merge: appendorder
|
||||||
|
normalize: dnd/spells
|
||||||
|
~~~
|
||||||
|
|
||||||
or object form:
|
| Field | Type | Default | Rules |
|
||||||
|
| --- | --- | --- | --- |
|
||||||
|
| **input** | module binding | none | Required. |
|
||||||
|
| **chunk** | module binding | **generic** | Optional. |
|
||||||
|
| **output** | module binding | **json** | Optional. |
|
||||||
|
| **artifacts** | map | none | Compact single-step lane map. |
|
||||||
|
| **steps** | list | none | Ordered lane definitions. Mutually exclusive with **artifacts**. |
|
||||||
|
| **references** | map | none | External reference defaults for eligible targets. |
|
||||||
|
|
||||||
```yaml
|
Use either **artifacts** or **steps**. The compact **artifacts** form is an
|
||||||
chunk:
|
implicit single step. An explicit **steps** list must be non-empty; every step
|
||||||
module: generic
|
needs a unique non-empty **id**, an **artifacts** map, and may have
|
||||||
llm_profile: default
|
**references**. A lane ID must not appear more than once in a pipeline,
|
||||||
|
including across explicit steps.
|
||||||
|
|
||||||
|
A lane has these fields:
|
||||||
|
|
||||||
|
| Field | Type | Default | Rules |
|
||||||
|
| --- | --- | --- | --- |
|
||||||
|
| **extract** | module binding | none | Required. |
|
||||||
|
| **merge** | module binding | **appendorder** | Optional. |
|
||||||
|
| **normalize** | module binding | **noop** | Optional. |
|
||||||
|
| **references** | map | none | Supported compatibility alias for **extract.references**. |
|
||||||
|
| **validators** | list | none | Non-empty lane-level lists are rejected. Set validator overrides on a binding instead. |
|
||||||
|
|
||||||
|
The lane-level **references** alias remains accepted. When the alias and
|
||||||
|
**extract.references** bind the same slot, **extract.references** wins. Use
|
||||||
|
the binding-local form in new configurations.
|
||||||
|
|
||||||
|
## Module Bindings And Validators
|
||||||
|
|
||||||
|
Use a module key directly when no other binding fields are needed:
|
||||||
|
|
||||||
|
~~~yaml
|
||||||
|
input: seriatim
|
||||||
|
~~~
|
||||||
|
|
||||||
|
Use an object for fields:
|
||||||
|
|
||||||
|
~~~yaml
|
||||||
|
extract:
|
||||||
|
module: dnd/spells
|
||||||
|
llm_profile: gemini-2-flash
|
||||||
|
retries: 2
|
||||||
|
references:
|
||||||
|
spell_catalog: ./dnd-spell-catalog.json
|
||||||
|
~~~
|
||||||
|
|
||||||
|
| Binding field | Type | Default | Rules |
|
||||||
|
| --- | --- | --- | --- |
|
||||||
|
| **module** | string | none | Required for an object binding. Must be a registered compatible key. |
|
||||||
|
| **llm_profile** | string | none | Optional non-empty Scriptorium profile ID. |
|
||||||
|
| **retries** | integer | 0 | Non-negative additional attempts for chunk, extract, merge, and normalize bindings. |
|
||||||
|
| **options** | object | none | Must satisfy the selected module. |
|
||||||
|
| **references** | map | none | Valid only on chunk, extract, merge, and normalize bindings. |
|
||||||
|
| **validators** | list | production chain | Valid only on chunk, extract, merge, and normalize bindings. |
|
||||||
|
|
||||||
|
Omitting **validators** uses the registered chain. **validators: []** selects
|
||||||
|
an empty chain; a non-empty list replaces the chain in the listed order.
|
||||||
|
Validator bindings accept only **module**, **llm_profile**, and **options**.
|
||||||
|
They reject **references**, **retries**, and nested **validators**. Deterministic
|
||||||
|
validators reject an explicit **llm_profile**.
|
||||||
|
|
||||||
|
The **json** output module accepts optional **include_chunk_map** and
|
||||||
|
**evidence_context** settings:
|
||||||
|
|
||||||
|
~~~yaml
|
||||||
|
output:
|
||||||
|
module: json
|
||||||
options:
|
options:
|
||||||
max_units: 50
|
include_chunk_map: true
|
||||||
```
|
evidence_context:
|
||||||
|
enabled: true
|
||||||
|
window_units: 3
|
||||||
|
lanes:
|
||||||
|
- npcs
|
||||||
|
- spells
|
||||||
|
~~~
|
||||||
|
|
||||||
Binding fields:
|
**include_chunk_map** is a boolean and defaults to false. It adds the accepted
|
||||||
|
chunk map when one exists; its wire format is defined in the
|
||||||
|
[chunk-map contract](integrations/chunk-map.md).
|
||||||
|
|
||||||
- `module`: module key.
|
Omitting **evidence_context** disables evidence publication. When present, it
|
||||||
- `llm_profile`: optional LLM profile ID. Empty means `default`.
|
is an object with these strict fields:
|
||||||
- `options`: optional module-specific settings.
|
|
||||||
|
|
||||||
The `--llm-profile` run flag overrides every effective module binding to use
|
| Field | Type | Rules |
|
||||||
one configured profile.
|
|
||||||
|
|
||||||
## Implemented Production Modules
|
|
||||||
|
|
||||||
| Slot | Key | Notes |
|
|
||||||
| --- | --- | --- |
|
| --- | --- | --- |
|
||||||
| input | `seriatim` | Reads Seriatim transcript JSON. |
|
| **enabled** | boolean | Required. `false` permits no other evidence fields. |
|
||||||
| chunk | `generic` | Splits source units into ordered chunks. |
|
| **lanes** | array of strings | Required and non-empty when enabled. Each value is trimmed and must be unique; every value must name a configured pipeline lane. |
|
||||||
| extract | `dnd/spells` | Extracts `dnd.spell_cast` artifacts. |
|
| **window_units** | non-negative integer | Optional when enabled; defaults to 3. Zero retains only directly cited units. |
|
||||||
| merge | `appendorder` | Keeps candidates in append order. |
|
|
||||||
| normalize | `noop` | Passes merged artifacts through unchanged. |
|
|
||||||
| output | `json` | Produces JSON output files. |
|
|
||||||
|
|
||||||
The `generic` chunker accepts:
|
Unknown outer or nested option fields are rejected, as are incompatible YAML
|
||||||
|
types. The allowlist remains valid when a run uses lane filtering: a configured
|
||||||
|
lane that is not active for that invocation simply contributes no evidence.
|
||||||
|
Evidence publication is opt-in because it can persist source text and metadata.
|
||||||
|
Its payload contract is [Published Evidence Context](integrations/evidence-context.md).
|
||||||
|
|
||||||
- `max_units`: positive integer, default `50`;
|
## References And Ordered Handoffs
|
||||||
- `overlap_units`: non-negative integer, default `0`, and must be less than
|
|
||||||
`max_units`.
|
|
||||||
|
|
||||||
## Diagnostics
|
Reference maps bind named slots that the selected target declares. A scalar is
|
||||||
|
an external path. Pipeline-level maps accept only external paths; step-local
|
||||||
|
and binding-local maps may also select a normalized artifact from an earlier
|
||||||
|
step:
|
||||||
|
|
||||||
`diagnostics` fields:
|
~~~yaml
|
||||||
|
steps:
|
||||||
|
- id: describe-session
|
||||||
|
artifacts:
|
||||||
|
npcs:
|
||||||
|
extract: dnd/npcs
|
||||||
|
normalize: dnd/npcs
|
||||||
|
- id: extract-events
|
||||||
|
references:
|
||||||
|
npcs:
|
||||||
|
artifact:
|
||||||
|
step: describe-session
|
||||||
|
lane: npcs
|
||||||
|
artifacts:
|
||||||
|
spells:
|
||||||
|
extract: dnd/spells
|
||||||
|
normalize: dnd/spells
|
||||||
|
~~~
|
||||||
|
|
||||||
- `work_dir`: directory for per-run diagnostics. Default: `/tmp/notarius`.
|
An artifact selector contains only **step** and **lane**. The producer must be
|
||||||
- `retention`: `auto`, `always`, or `never`. Empty uses `auto`.
|
an earlier step and the selected artifact must be compatible with the consumer
|
||||||
|
slot. A generated binding supplies one accepted normalized artifact; it does
|
||||||
|
not name a file. A configured generated dependency remains required even when
|
||||||
|
that consumer slot is otherwise optional.
|
||||||
|
|
||||||
`auto` retains diagnostics for failed runs and successful runs with warnings.
|
Pipeline references are defaults. A matching step-local or binding-local
|
||||||
`always` retains diagnostics for every run. `never` removes diagnostics for
|
external path overrides a pipeline default. Required slots must be bound after
|
||||||
successful runs without regard to warnings; failed runs are retained.
|
these configuration values and any CLI reference overrides are applied.
|
||||||
|
Reference paths in YAML are resolved relative to the configuration file.
|
||||||
|
|
||||||
The `--diagnostics-dir` run flag overrides `diagnostics.work_dir` for that
|
### D&D Reference Slots
|
||||||
invocation.
|
|
||||||
|
The following slot names are accepted by the implemented D&D modules when the
|
||||||
|
selected target declares them:
|
||||||
|
|
||||||
|
| Slot | Source and use |
|
||||||
|
| --- | --- |
|
||||||
|
| **party** | Optional text campaign context. This is the canonical party-roster spelling. |
|
||||||
|
| **roster** | Accepted compatibility alias for **party**. |
|
||||||
|
| **players** | Optional text player context. |
|
||||||
|
| **glossary** | Optional text campaign glossary. |
|
||||||
|
| **spell_catalog** | Optional JSON spell-catalog overlay for spell extraction and normalization. See [spell-catalog overlays](integrations/dnd-spell-catalog-overlays.md). |
|
||||||
|
| **npcs** | Normalized NPC registry. Optional for spells and combat turns; required for NPC interactions. |
|
||||||
|
| **scene_descriptions** | Required normalized scene-description artifact for combat-turn extraction. |
|
||||||
|
|
||||||
|
Scene descriptions accept **party**, **players**, and **glossary**, but not
|
||||||
|
**roster**. NPC interactions require **npcs** for both extraction and
|
||||||
|
normalization. Combat turns require **scene_descriptions** for extraction; the
|
||||||
|
normalized combat-turn module may use optional **npcs**. The complete example
|
||||||
|
shows generated **npcs** and **scene_descriptions** bindings.
|
||||||
|
|
||||||
|
## Production Module Keys
|
||||||
|
|
||||||
|
| Kind | Keys |
|
||||||
|
| --- | --- |
|
||||||
|
| Input | **seriatim** |
|
||||||
|
| Chunk | **generic**, **dnd/scenes** |
|
||||||
|
| Extract | **dnd/spells**, **dnd/npcs**, **dnd/combat-turns**, **dnd/item-events**, **dnd/npc-interactions**, **dnd/scene-descriptions** |
|
||||||
|
| Merge | **appendorder** |
|
||||||
|
| Normalize | **noop**, **dnd/spells**, **dnd/npcs**, **dnd/combat-turns**, **dnd/item-events**, **dnd/npc-interactions**, **dnd/scene-descriptions** |
|
||||||
|
| Output | **json** |
|
||||||
|
|
||||||
|
The D&D artifact contracts define each emitted schema:
|
||||||
|
[spells](integrations/dnd-spell-artifacts.md),
|
||||||
|
[NPCs](integrations/dnd-npc-artifacts.md),
|
||||||
|
[NPC interactions](integrations/dnd-npc-interaction-artifacts.md),
|
||||||
|
[combat turns](integrations/dnd-combat-turn-artifacts.md),
|
||||||
|
[item events](integrations/dnd-item-event-artifacts.md), and
|
||||||
|
[scene descriptions](integrations/dnd-scene-description-artifacts.md).
|
||||||
|
|
||||||
|
## Production Validator Keys And Default Chains
|
||||||
|
|
||||||
|
Available validator keys are:
|
||||||
|
|
||||||
|
| Family | Keys |
|
||||||
|
| --- | --- |
|
||||||
|
| Generic | **generic/always_accept**, **generic/always_reject**, **generic/valid_json**, **generic/valid_json_schema** |
|
||||||
|
| Spells | **extract/dnd/spells/shape**, **extract/dnd/spells/catalog**, **extract/dnd/spells/source_refs**, **extract/dnd/spells/source_relatedness** |
|
||||||
|
| NPCs | **extract/dnd/npcs/shape**, **extract/dnd/npcs/source_refs**, **extract/dnd/npcs/source_relatedness**, **normalize/dnd/npcs/identity** |
|
||||||
|
| Combat turns | **extract/dnd/combat-turns/shape**, **extract/dnd/combat-turns/source_refs**, **extract/dnd/combat-turns/source_relatedness**, **normalize/dnd/combat-turns/invariants** |
|
||||||
|
| Item events | **extract/dnd/item-events/shape**, **extract/dnd/item-events/source_refs**, **extract/dnd/item-events/source_relatedness**, **normalize/dnd/item-events/invariants** |
|
||||||
|
| NPC interactions | **extract/dnd/npc-interactions/shape**, **extract/dnd/npc-interactions/registry**, **extract/dnd/npc-interactions/source_refs**, **extract/dnd/npc-interactions/source_relatedness**, **normalize/dnd/npc-interactions/invariants** |
|
||||||
|
| Scene descriptions | **extract/dnd/scene-descriptions/shape**, **extract/dnd/scene-descriptions/source_refs**, **extract/dnd/scene-descriptions/source_relatedness**, **normalize/dnd/scene-descriptions/invariants** |
|
||||||
|
|
||||||
|
When no override is configured, production D&D bindings use the following
|
||||||
|
ordered chains. Each row lists extract then normalize; spell chains are the
|
||||||
|
same at both stages.
|
||||||
|
|
||||||
|
| Lane | Extract | Normalize |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| Spells | generic/valid_json, extract/dnd/spells/shape, extract/dnd/spells/catalog, extract/dnd/spells/source_refs, generic/valid_json_schema, extract/dnd/spells/source_relatedness | Same as extract |
|
||||||
|
| NPCs | generic/valid_json, extract/dnd/npcs/shape, extract/dnd/npcs/source_refs, generic/valid_json_schema, extract/dnd/npcs/source_relatedness | generic/valid_json, extract/dnd/npcs/shape, normalize/dnd/npcs/identity, extract/dnd/npcs/source_refs, generic/valid_json_schema, extract/dnd/npcs/source_relatedness |
|
||||||
|
| Combat turns | generic/valid_json, extract/dnd/combat-turns/shape, extract/dnd/combat-turns/source_refs, generic/valid_json_schema, extract/dnd/combat-turns/source_relatedness | generic/valid_json, extract/dnd/combat-turns/shape, normalize/dnd/combat-turns/invariants, extract/dnd/combat-turns/source_refs, generic/valid_json_schema, extract/dnd/combat-turns/source_relatedness |
|
||||||
|
| Item events | generic/valid_json, extract/dnd/item-events/shape, extract/dnd/item-events/source_refs, generic/valid_json_schema, extract/dnd/item-events/source_relatedness | generic/valid_json, extract/dnd/item-events/shape, normalize/dnd/item-events/invariants, extract/dnd/item-events/source_refs, generic/valid_json_schema, extract/dnd/item-events/source_relatedness |
|
||||||
|
| NPC interactions | generic/valid_json, extract/dnd/npc-interactions/shape, extract/dnd/npc-interactions/registry, extract/dnd/npc-interactions/source_refs, generic/valid_json_schema, extract/dnd/npc-interactions/source_relatedness | generic/valid_json, extract/dnd/npc-interactions/shape, extract/dnd/npc-interactions/registry, normalize/dnd/npc-interactions/invariants, extract/dnd/npc-interactions/source_refs, generic/valid_json_schema, extract/dnd/npc-interactions/source_relatedness |
|
||||||
|
| Scene descriptions | generic/valid_json, extract/dnd/scene-descriptions/shape, extract/dnd/scene-descriptions/source_refs, generic/valid_json_schema, extract/dnd/scene-descriptions/source_relatedness | generic/valid_json, extract/dnd/scene-descriptions/shape, normalize/dnd/scene-descriptions/invariants, extract/dnd/scene-descriptions/source_refs, generic/valid_json_schema, extract/dnd/scene-descriptions/source_relatedness |
|
||||||
|
|
||||||
|
Chains are only registered for the D&D extract and normalize modules shown
|
||||||
|
above; select an explicit override when a different compatible chain is
|
||||||
|
required.
|
||||||
|
|
||||||
## Validation
|
## Validation
|
||||||
|
|
||||||
Configuration validation checks:
|
Validate a file and one pipeline before running it:
|
||||||
|
|
||||||
- supported config version and known YAML fields;
|
~~~sh
|
||||||
- non-empty, non-duplicated IDs after trimming;
|
go run ./cmd/notarius config validate \
|
||||||
- supported LLM provider and non-negative profile limits;
|
--config examples/dnd-minimal.config.yml \
|
||||||
- positive global LLM concurrency;
|
--pipeline dnd-session
|
||||||
- supported diagnostics retention and non-empty work directory;
|
~~~
|
||||||
- module binding LLM profiles refer to configured profiles.
|
|
||||||
|
|
||||||
Pipeline resolution additionally checks:
|
Configuration validation rejects invalid YAML, unsupported fields, invalid
|
||||||
|
defaults or environment overrides, incompatible module keys, unknown options,
|
||||||
- the pipeline ID exists;
|
invalid reference bindings, missing required reference slots, invalid validator
|
||||||
- at least one artifact lane is declared and selected;
|
overrides, and incompatible generated artifact handoffs. Use
|
||||||
- selected lanes exist when `--only` is used;
|
[pipelines list](cli.md#pipelines-list) to inspect configured IDs.
|
||||||
- required module keys are present;
|
|
||||||
- module keys are registered for the expected slot;
|
|
||||||
- module capability requirements are satisfied.
|
|
||||||
|
|||||||
74
docs/consumers/subprocess.md
Normal file
74
docs/consumers/subprocess.md
Normal file
@@ -0,0 +1,74 @@
|
|||||||
|
# Using Notarius As A Subprocess
|
||||||
|
|
||||||
|
Use this workflow when an orchestrator runs Notarius and consumes its published
|
||||||
|
artifacts. The [CLI reference](../cli.md) owns invocation syntax and exit
|
||||||
|
statuses, while the [run-result receipt](../integrations/run-result.md) and
|
||||||
|
[Published JSON Output contract](../integrations/json-output.md) own the
|
||||||
|
durable result formats.
|
||||||
|
|
||||||
|
## Run And Check The Process
|
||||||
|
|
||||||
|
Optionally preflight a selected configuration and pipeline before work starts:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
notarius config validate --config /path/to/notarius.yml --pipeline pipeline-id
|
||||||
|
```
|
||||||
|
|
||||||
|
Invoke the run with explicit paths and machine-readable output. Capture
|
||||||
|
standard output and standard error separately; do not combine them before
|
||||||
|
processing the result.
|
||||||
|
|
||||||
|
```sh
|
||||||
|
notarius run pipeline-id \
|
||||||
|
--config /path/to/notarius.yml \
|
||||||
|
--input /path/to/source.json \
|
||||||
|
--output-dir /path/to/output-root \
|
||||||
|
--json
|
||||||
|
```
|
||||||
|
|
||||||
|
Use absolute paths for supplied input, configuration, output-root, and
|
||||||
|
reference files. When a stable prompt session identifier or references are
|
||||||
|
needed, pass the supported CLI flags. Supply credentials through Notarius's
|
||||||
|
documented configuration and environment mechanisms, never as command-line
|
||||||
|
arguments or generated secret-bearing configuration.
|
||||||
|
|
||||||
|
Wait for the process before interpreting standard output. Only an exit status
|
||||||
|
of 0 permits decoding the receipt. On a nonzero exit, retain standard error for
|
||||||
|
diagnosis and ignore all standard-output bytes: a failed receipt write may have
|
||||||
|
left a partial document.
|
||||||
|
|
||||||
|
## Discover Required Artifacts
|
||||||
|
|
||||||
|
Decode the successful receipt and accept the schema versions supported by the
|
||||||
|
caller. Use its `output_directory` as the bundle root. For the production JSON
|
||||||
|
output, resolve `index_file` under that root with a confinement check and reject
|
||||||
|
an absolute path or a result that escapes the root.
|
||||||
|
|
||||||
|
Read the resulting `index.json` and locate each artifact by `lane_id`, not by a
|
||||||
|
guessed filename. Before decoding a selected payload, verify its descriptor's
|
||||||
|
media type and schema identity against the relevant published artifact
|
||||||
|
contract. The JSON bundle contract links to the available lane contracts.
|
||||||
|
|
||||||
|
If `index.json` has an `evidence_context` descriptor, treat it as a
|
||||||
|
pipeline-wide artifact rather than a lane entry. Verify its six descriptor
|
||||||
|
fields before decoding the linked file according to the [Published Evidence
|
||||||
|
Context contract](../integrations/evidence-context.md). Use each
|
||||||
|
`evidence_refs` entry as the citation to source material. Its surrounding
|
||||||
|
context range and included units explain the citation, but do not widen or
|
||||||
|
replace the cited source reference.
|
||||||
|
|
||||||
|
A zero exit status may still report rejected outputs, warnings, or absent
|
||||||
|
lanes. The caller decides which lane IDs are required for its own work and
|
||||||
|
which are optional; it should make that decision explicitly rather than infer
|
||||||
|
failure from the receipt counts alone.
|
||||||
|
|
||||||
|
## Preserve Provenance And Handle Data Carefully
|
||||||
|
|
||||||
|
Keep the receipt with the published `manifest.json`, and retain
|
||||||
|
`rejected.json` and `warnings.json` when review or later provenance requires
|
||||||
|
them. Treat the input, output bundle, cache, debug bundle, and captured process
|
||||||
|
logs as potentially sensitive data. Apply the caller's access controls and
|
||||||
|
retention policy, and avoid copying secrets into arguments, logs, or
|
||||||
|
provenance records. An evidence-context artifact contains source-unit text and
|
||||||
|
metadata, and selected lanes can cover most of an input; preserve and share it
|
||||||
|
only when that source content is authorized for the recipient.
|
||||||
43
docs/development.md
Normal file
43
docs/development.md
Normal file
@@ -0,0 +1,43 @@
|
|||||||
|
# Development
|
||||||
|
|
||||||
|
This is the first-read landing page for people and LLM coding agents working on
|
||||||
|
Notarius. It provides a concise repository orientation and routes each kind of
|
||||||
|
change to its canonical documentation.
|
||||||
|
|
||||||
|
Notarius is a Go CLI for configured structured extraction workflows. Start with
|
||||||
|
the [README](../README.md) for product context, [Architecture](policy/architecture.md)
|
||||||
|
for system boundaries, and [Internal Overview](internal/overview.md) for the
|
||||||
|
implemented component map.
|
||||||
|
|
||||||
|
## What To Read
|
||||||
|
|
||||||
|
| When working on | Read | Why |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| Finding the package or component that owns current behavior | [Internal Overview](internal/overview.md) | It is the implemented component inventory and routes to focused internals. |
|
||||||
|
| Application shape, package boundaries, contracts, dependency direction, runtime guarantees, or safety properties | [Architecture](policy/architecture.md) and relevant [ADRs](adr/) | Architecture defines the intended system and its invariants; ADRs preserve significant decision rationale. |
|
||||||
|
| Any documentation addition or revision | [Documentation Policy](policy/documentation.md) | It defines canonical homes, audiences, current-behavior rules, and maintenance requirements. |
|
||||||
|
| Adding, changing, reviewing, or deleting tests | [Testing Policy](policy/testing.md) | It defines risk-based sufficiency, durable test boundaries, test-double guidance, and criteria for retaining tests. |
|
||||||
|
| CLI composition or command behavior | [CLI Internals](internal/cli.md) and [CLI Reference](cli.md) | The internal guide owns composition and command flow; the reference owns public syntax. |
|
||||||
|
| Building a subprocess caller or changing its result protocol | [Subprocess Consumer Guide](consumers/subprocess.md), [Run Result Receipt](integrations/run-result.md), and [CLI Internals](internal/cli.md) | These separate caller workflow, durable receipt contract, and CLI implementation behavior. |
|
||||||
|
| Configuration loading, resolution, or user-visible configuration behavior | [Configuration Internals](internal/configuration.md) and [Configuration](config.md) | The internal guide owns loading and resolution mechanics; the reference owns the configuration contract. |
|
||||||
|
| Pipeline resolution or execution | [Pipeline Internals](internal/pipeline.md) | It documents profiles, references, validation, retries, checkpoints, and runner behavior. |
|
||||||
|
| Production modules or validators | [Module Internals](internal/modules.md), [D&D Module Internals](internal/dnd.md), and [D&D integration contracts](integrations/) | The generic guide owns extension mechanics, the D&D guide owns shared family conventions, and the contracts own durable output shapes. |
|
||||||
|
| LLM clients, prompts, schemas, profiles, or scheduling | [LLM Runtime](internal/llm.md) | It documents the transport boundary and Scriptorium integration. |
|
||||||
|
| Output, cache, resume, or debug artifacts | [Run State Internals](internal/state.md), [Operations](operations.md), and [Configuration](config.md) | These separate implementation details, operator behavior, and configuration contracts. |
|
||||||
|
| External input formats, artifact schemas, or durable output files | [Integration Contracts](integrations/) | Integration documents define external and durable data contracts. |
|
||||||
|
| Proposed or unimplemented behavior | [Roadmap](roadmap/) | Future work belongs only in roadmap documentation until implemented. |
|
||||||
|
|
||||||
|
For an existing subsystem, also inspect its focused tests and the package-local
|
||||||
|
types and contracts before changing behavior.
|
||||||
|
|
||||||
|
## Validation
|
||||||
|
|
||||||
|
Use focused package tests while iterating. Run the repository-wide checks when
|
||||||
|
a change affects shared contracts, application behavior, or maintained
|
||||||
|
documentation examples:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
go test ./...
|
||||||
|
go vet ./...
|
||||||
|
go build ./cmd/notarius
|
||||||
|
```
|
||||||
81
docs/integrations/chunk-map.md
Normal file
81
docs/integrations/chunk-map.md
Normal file
@@ -0,0 +1,81 @@
|
|||||||
|
# Accepted Chunk Map
|
||||||
|
|
||||||
|
This document defines the optional durable `chunk-map.json` artifact in a
|
||||||
|
[published JSON bundle](json-output.md). It describes the accepted,
|
||||||
|
materialized chunk plan used by one run. It is not a lane payload and is never
|
||||||
|
an input to a later pipeline step.
|
||||||
|
|
||||||
|
## Contract Identity
|
||||||
|
|
||||||
|
| Property | Value |
|
||||||
|
| --- | --- |
|
||||||
|
| Artifact kind | `source/chunk-map` |
|
||||||
|
| Logical file | `chunk-map.json` |
|
||||||
|
| Media type | `application/json` |
|
||||||
|
| Schema ID | `notarius.source.chunk_map` |
|
||||||
|
| Schema name | `notarius_source_chunk_map_v1` |
|
||||||
|
| Schema version | `v1` |
|
||||||
|
|
||||||
|
The optional `chunk_map` descriptor in `index.json` identifies this artifact.
|
||||||
|
Export is controlled by the JSON output binding described in
|
||||||
|
[Configuration](../config.md#module-bindings-and-validators).
|
||||||
|
|
||||||
|
## Wire Shape
|
||||||
|
|
||||||
|
Every payload has these required fields:
|
||||||
|
|
||||||
|
| Field | Meaning |
|
||||||
|
| --- | --- |
|
||||||
|
| `source_id` | Accepted source-document identity. |
|
||||||
|
| `source_digest` | Lower-case `sha256:` digest of that source document. |
|
||||||
|
| `plan_digest` | Lower-case `sha256:` digest of the logical chunk plan. |
|
||||||
|
| `requested_chunker` | Chunk module selected by the resolved pipeline. |
|
||||||
|
| `producer` | Original accepted-plan producer. `input_module` and `chunk_module` are required; `llm_profile` is optional. |
|
||||||
|
| `plan_annotations` | Plan-level annotation namespace map; `{}` when none are present. |
|
||||||
|
| `chunks` | Non-empty execution-order chunk collection. |
|
||||||
|
|
||||||
|
Each `chunks` entry contains non-empty `id`, zero-based `index`, `source_ref`,
|
||||||
|
positive `unit_count`, and an explicit `annotations` map. `source_ref` contains
|
||||||
|
the same `source_id` as the top-level value plus positive inclusive
|
||||||
|
`start_unit_id` and `end_unit_id` values. Endpoints identify source units; their
|
||||||
|
numeric values do not by themselves establish source-document order.
|
||||||
|
|
||||||
|
Annotation namespaces are non-empty trimmed strings. Their values are arbitrary
|
||||||
|
valid JSON and are retained without interpreting a module-specific namespace.
|
||||||
|
|
||||||
|
## Ordering And Validation
|
||||||
|
|
||||||
|
`chunks` are in execution order. Their indexes are contiguous, start at zero,
|
||||||
|
and equal their array positions; chunk IDs are unique. The emitted map is built
|
||||||
|
only after the selected plan has been accepted and materialized against the
|
||||||
|
source document, so its ranges, unit counts, annotations, and digests describe
|
||||||
|
that exact plan.
|
||||||
|
|
||||||
|
The codec rejects malformed JSON, trailing content, unknown fixed-object
|
||||||
|
fields, invalid identities or digests, invalid annotations, duplicate chunk
|
||||||
|
IDs, non-contiguous indexes, and a `plan_digest` that does not match the
|
||||||
|
reconstructed logical plan. The checked-in
|
||||||
|
[schema](../../internal/framework/chunkmap/assets/schemas/source_chunk_map.v1.json)
|
||||||
|
defines the strict JSON shape.
|
||||||
|
|
||||||
|
## Valid Example
|
||||||
|
|
||||||
|
The compact
|
||||||
|
[source chunk-map fixture](../../internal/framework/chunkmap/testdata/source_chunk_map.v1.json)
|
||||||
|
is decoded by the production codec and demonstrates an accepted map with
|
||||||
|
annotations, producer identity, and ordered chunks.
|
||||||
|
|
||||||
|
## Publication And Compatibility
|
||||||
|
|
||||||
|
The map is present only when a chunk plan was accepted and its export is
|
||||||
|
enabled. It remains publishable if a later lane is rejected, but is absent when
|
||||||
|
chunk-plan validation rejects the plan. `requested_chunker` identifies the
|
||||||
|
current pipeline selection, while `producer` identifies the component that
|
||||||
|
originally produced the accepted plan; they may differ when an accepted plan is
|
||||||
|
reused.
|
||||||
|
|
||||||
|
The map contains structure rather than source content: it excludes transcript
|
||||||
|
bytes, source-unit metadata, chunk text, private model output, reference
|
||||||
|
content, debug data, and filesystem paths. Treat the exported map with the
|
||||||
|
same care as other published output. Publication location and retention are
|
||||||
|
defined in [Operations](../operations.md#output-bundles).
|
||||||
69
docs/integrations/dnd-combat-turn-artifacts.md
Normal file
69
docs/integrations/dnd-combat-turn-artifacts.md
Normal file
@@ -0,0 +1,69 @@
|
|||||||
|
# D&D Combat-Turn Artifact
|
||||||
|
|
||||||
|
This contract defines the durable combat-action occurrence list produced by
|
||||||
|
`dnd/combat-turns`. It records source-grounded turns and actions; it is not a
|
||||||
|
complete initiative tracker, combat summary, or state model.
|
||||||
|
|
||||||
|
## Identity and compatibility
|
||||||
|
|
||||||
|
| Property | Value |
|
||||||
|
| --- | --- |
|
||||||
|
| Artifact kind | `dnd/combat-turn-list` |
|
||||||
|
| Schema ID | `notarius.dnd.combat_turns` |
|
||||||
|
| Schema name | `notarius_dnd_combat_turns_v1` |
|
||||||
|
| Schema version | `v1` |
|
||||||
|
| Media type | `application/json` |
|
||||||
|
|
||||||
|
`v1` is a strict JSON object with required `combat_turns`; the array may be
|
||||||
|
empty. Turn and source-reference objects reject unknown fields. An incompatible
|
||||||
|
shape change requires a new schema version.
|
||||||
|
|
||||||
|
## Wire shape
|
||||||
|
|
||||||
|
Each combat turn has these required fields:
|
||||||
|
|
||||||
|
| Field | Contract |
|
||||||
|
| --- | --- |
|
||||||
|
| `actor` | Non-empty acting character or creature name. |
|
||||||
|
| `turn_kind` | `turn`, `reaction`, `legendary_action`, `lair_action`, or `other`. |
|
||||||
|
| `source_refs` | One or more transcript evidence ranges. |
|
||||||
|
|
||||||
|
Each source reference has exactly `source_id`, `start_unit_id`, and
|
||||||
|
`end_unit_id`. It identifies an inclusive current-transcript range; unit IDs
|
||||||
|
are positive and the start may not follow the end.
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"combat_turns": [
|
||||||
|
{
|
||||||
|
"actor": "Mira Thorn",
|
||||||
|
"turn_kind": "turn",
|
||||||
|
"source_refs": [
|
||||||
|
{"source_id": "session-7", "start_unit_id": 31, "end_unit_id": 32}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
## Eligibility, evidence, and normalized form
|
||||||
|
|
||||||
|
The extractor requires an approved [scene-description artifact](dnd-scene-description-artifacts.md).
|
||||||
|
It emits combat turns only for a chunk with an exact matching scene classified
|
||||||
|
`combat`; an exact non-combat scene produces an accepted empty list. The scene
|
||||||
|
record controls eligibility only: its title, summary, and reference do not
|
||||||
|
become turn evidence. No exact matching scene also produces an empty list and
|
||||||
|
the `scene_classification_unavailable` warning.
|
||||||
|
|
||||||
|
An optional normalized [NPC artifact](dnd-npc-artifacts.md) can ground an
|
||||||
|
actor name. Its registry references are provenance, never combat evidence.
|
||||||
|
Normalization trims and, where possible, canonicalizes actor names; orders and
|
||||||
|
deduplicates exact source references; orders valid-evidence turns by source
|
||||||
|
chronology; and collapses only duplicates with the same actor identity, turn
|
||||||
|
kind, and complete valid evidence. It does not infer turns, initiative, or
|
||||||
|
actions from registry or scene data.
|
||||||
|
|
||||||
|
The [NPC-interaction artifact](dnd-npc-interaction-artifacts.md) records
|
||||||
|
broader NPC occurrences. The [JSON output contract](json-output.md) defines
|
||||||
|
publication, and [D&D module internals](../internal/dnd.md) describes routing
|
||||||
|
and validation mechanics.
|
||||||
78
docs/integrations/dnd-item-event-artifacts.md
Normal file
78
docs/integrations/dnd-item-event-artifacts.md
Normal file
@@ -0,0 +1,78 @@
|
|||||||
|
# D&D Item-Event Artifact
|
||||||
|
|
||||||
|
This contract defines the durable item and currency occurrence list produced by
|
||||||
|
`dnd/item-events`. It records source-grounded discoveries and possession
|
||||||
|
changes; it does not maintain an inventory, balance, or ledger.
|
||||||
|
|
||||||
|
## Identity and compatibility
|
||||||
|
|
||||||
|
| Property | Value |
|
||||||
|
| --- | --- |
|
||||||
|
| Artifact kind | `dnd/item-event-list` |
|
||||||
|
| Schema ID | `notarius.dnd.item_events` |
|
||||||
|
| Schema name | `notarius_dnd_item_events_v1` |
|
||||||
|
| Schema version | `v1` |
|
||||||
|
| Media type | `application/json` |
|
||||||
|
|
||||||
|
`v1` is a strict JSON object with required `events`; the array may be empty.
|
||||||
|
Event and source-reference objects reject unknown fields. An incompatible
|
||||||
|
shape change requires a new schema version.
|
||||||
|
|
||||||
|
## Wire shape
|
||||||
|
|
||||||
|
Every event has required `name`, `kind`, and `source_refs`. `quantity`, `from`,
|
||||||
|
and `to` are optional where the event kind permits them.
|
||||||
|
|
||||||
|
| Field | Contract |
|
||||||
|
| --- | --- |
|
||||||
|
| `name` | Non-empty item or currency display name. |
|
||||||
|
| `kind` | `discovered`, `acquired`, `lost`, `consumed`, or `transferred`. |
|
||||||
|
| `quantity` | Optional positive integer; omit it when no count is established. |
|
||||||
|
| `from` | Optional non-empty losing holder, when allowed by `kind`. |
|
||||||
|
| `to` | Optional non-empty gaining holder, when allowed by `kind`. |
|
||||||
|
| `source_refs` | One or more transcript evidence ranges. |
|
||||||
|
|
||||||
|
Each source reference has exactly `source_id`, `start_unit_id`, and
|
||||||
|
`end_unit_id`. It identifies an inclusive current-transcript range; unit IDs
|
||||||
|
are positive and the start may not follow the end.
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"events": [
|
||||||
|
{
|
||||||
|
"name": "Silver Pieces",
|
||||||
|
"kind": "acquired",
|
||||||
|
"quantity": 20,
|
||||||
|
"to": "party",
|
||||||
|
"source_refs": [
|
||||||
|
{"source_id": "session-7", "start_unit_id": 2, "end_unit_id": 2}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
## Holder rules and minimal extraction
|
||||||
|
|
||||||
|
`discovered` has neither holder; `acquired` requires `to` and forbids `from`;
|
||||||
|
`lost` and `consumed` require `from` and forbid `to`; `transferred` requires
|
||||||
|
both holders. `party` denotes collective possession. A transfer cannot use
|
||||||
|
`party` for either holder and its two normalized holders must differ.
|
||||||
|
|
||||||
|
Only an evidenced discovery or possession change belongs in this artifact.
|
||||||
|
It does not infer quantities or holders, convert currency denominations,
|
||||||
|
calculate balances, or merge nearby events. Campaign references may
|
||||||
|
disambiguate names but are never event evidence. Currency uses the ordinary
|
||||||
|
`name` field and an explicit `quantity` only when the transcript establishes
|
||||||
|
one; each denomination remains a separate event.
|
||||||
|
|
||||||
|
Normalization trims display whitespace, orders and removes exact duplicate
|
||||||
|
source references, then orders events by valid source chronology, name identity
|
||||||
|
and display value, kind, holders, quantity, and reference sequence. It
|
||||||
|
collapses only entries with the same normalized durable fields and complete
|
||||||
|
valid evidence.
|
||||||
|
|
||||||
|
The [JSON output contract](json-output.md) defines publication. See
|
||||||
|
[D&D module internals](../internal/dnd.md) for implementation details and the
|
||||||
|
[NPC-interaction artifact](dnd-npc-interaction-artifacts.md) for a distinct
|
||||||
|
kind of occurrence.
|
||||||
69
docs/integrations/dnd-npc-artifacts.md
Normal file
69
docs/integrations/dnd-npc-artifacts.md
Normal file
@@ -0,0 +1,69 @@
|
|||||||
|
# D&D NPC Artifact
|
||||||
|
|
||||||
|
This contract defines the durable NPC registry produced by `dnd/npcs`. It is a
|
||||||
|
minimal, source-grounded identity registry for other D&D artifacts, not a
|
||||||
|
character sheet or a relationship summary.
|
||||||
|
|
||||||
|
## Identity and compatibility
|
||||||
|
|
||||||
|
| Property | Value |
|
||||||
|
| --- | --- |
|
||||||
|
| Artifact kind | `dnd/npc-list` |
|
||||||
|
| Schema ID | `notarius.dnd.npcs` |
|
||||||
|
| Schema name | `notarius_dnd_npcs_v1` |
|
||||||
|
| Schema version | `v1` |
|
||||||
|
| Media type | `application/json` |
|
||||||
|
| Identity policy | `dnd.npcs.identity.v1` |
|
||||||
|
|
||||||
|
`v1` accepts one strict JSON object with required `npcs`; the array may be
|
||||||
|
empty. NPC and source-reference objects reject unknown fields. An incompatible
|
||||||
|
artifact shape or identity-policy change uses a new version or policy.
|
||||||
|
|
||||||
|
## Wire shape and identity
|
||||||
|
|
||||||
|
Each NPC has these required fields:
|
||||||
|
|
||||||
|
| Field | Contract |
|
||||||
|
| --- | --- |
|
||||||
|
| `id` | `npc:sha256:` followed by 64 lowercase hexadecimal characters. |
|
||||||
|
| `name` | Non-empty canonical display name. |
|
||||||
|
| `source_refs` | One or more transcript evidence ranges for the identity. |
|
||||||
|
|
||||||
|
A source reference has exactly `source_id`, `start_unit_id`, and `end_unit_id`.
|
||||||
|
The source ID identifies the transcript, unit IDs are positive inclusive unit
|
||||||
|
identifiers, and the start may not follow the end.
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"npcs": [
|
||||||
|
{
|
||||||
|
"id": "npc:sha256:99a16589618a04f535a7d21fdcc71a0b1c05d22f752cd492065b1086d97bc3d7",
|
||||||
|
"name": "Mira Thorn",
|
||||||
|
"source_refs": [
|
||||||
|
{"source_id": "session-7", "start_unit_id": 4, "end_unit_id": 5}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
The ID is deterministic: normalize the name to Unicode NFKC, normalize the
|
||||||
|
supported apostrophe forms, collapse whitespace, case-fold it, SHA-256 the
|
||||||
|
result, then prefix the lowercase hexadecimal digest with `npc:sha256:`. Each
|
||||||
|
canonical identity and ID appears at most once. Normalization collapses records
|
||||||
|
with the same canonical identity, retains their earliest position, and merges
|
||||||
|
their canonicalized evidence; it does not add aliases, roles, descriptions, or
|
||||||
|
relationship fields.
|
||||||
|
|
||||||
|
## Scope and consumers
|
||||||
|
|
||||||
|
Only individually identifiable NPC names with transcript evidence belong in
|
||||||
|
this artifact. Groups, generic roles, invented labels, and descriptive
|
||||||
|
enrichment are excluded. Its source references prove registry provenance; they
|
||||||
|
do not become evidence for a spell, interaction, or combat occurrence.
|
||||||
|
|
||||||
|
This registry can ground actor or caster names in the [spell](dnd-spell-artifacts.md)
|
||||||
|
and [combat-turn](dnd-combat-turn-artifacts.md) artifacts. It is required to
|
||||||
|
resolve the canonical `name` in an [NPC interaction](dnd-npc-interaction-artifacts.md).
|
||||||
|
The [JSON output contract](json-output.md) defines publication, and
|
||||||
|
[D&D module internals](../internal/dnd.md) owns pipeline mechanics.
|
||||||
78
docs/integrations/dnd-npc-interaction-artifacts.md
Normal file
78
docs/integrations/dnd-npc-interaction-artifacts.md
Normal file
@@ -0,0 +1,78 @@
|
|||||||
|
# D&D NPC Interaction Artifact
|
||||||
|
|
||||||
|
This contract defines the durable occurrence list produced by
|
||||||
|
`dnd/npc-interactions`. It records discrete, source-grounded interactions with
|
||||||
|
NPCs already present in a normalized registry; it does not extend that registry
|
||||||
|
or summarize the session.
|
||||||
|
|
||||||
|
## Identity and compatibility
|
||||||
|
|
||||||
|
| Property | Value |
|
||||||
|
| --- | --- |
|
||||||
|
| Artifact kind | `dnd/npc-interaction-list` |
|
||||||
|
| Schema ID | `notarius.dnd.npc_interactions` |
|
||||||
|
| Schema name | `notarius_dnd_npc_interactions_v1` |
|
||||||
|
| Schema version | `v1` |
|
||||||
|
| Media type | `application/json` |
|
||||||
|
|
||||||
|
`v1` is a strict JSON object with required `interactions`; the array may be
|
||||||
|
empty. Interaction and source-reference objects reject unknown fields. An
|
||||||
|
incompatible shape change requires a new schema version.
|
||||||
|
|
||||||
|
## Wire shape
|
||||||
|
|
||||||
|
Each interaction has these required fields:
|
||||||
|
|
||||||
|
| Field | Contract |
|
||||||
|
| --- | --- |
|
||||||
|
| `name` | Non-empty canonical display name from the required NPC registry. |
|
||||||
|
| `kind` | One of the interaction categories below. |
|
||||||
|
| `source_refs` | One or more transcript evidence ranges. |
|
||||||
|
|
||||||
|
Each source reference has exactly `source_id`, `start_unit_id`, and
|
||||||
|
`end_unit_id`. It identifies an inclusive range in the current transcript;
|
||||||
|
unit IDs are positive and the start may not follow the end. Extraction evidence
|
||||||
|
for an interaction is confined to its accepted chunk.
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"interactions": [
|
||||||
|
{
|
||||||
|
"name": "Mira Thorn",
|
||||||
|
"kind": "dialogue",
|
||||||
|
"source_refs": [
|
||||||
|
{"source_id": "session-7", "start_unit_id": 12, "end_unit_id": 13}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
## Interaction categories
|
||||||
|
|
||||||
|
| Kind | Meaning |
|
||||||
|
| --- | --- |
|
||||||
|
| `mentioned` | The NPC is referred to but is not established as present or communicating. |
|
||||||
|
| `noncombat_presence` | The NPC is present and relevant without meaningful dialogue or combat participation. |
|
||||||
|
| `dialogue` | The NPC speaks, responds, or meaningfully participates in a non-combat exchange. |
|
||||||
|
| `combat_ally` | The NPC actively participates in combat on the party's side. |
|
||||||
|
| `combat_opponent` | The NPC actively participates in combat against the party. |
|
||||||
|
| `other` | A clearly evidenced direct occurrence not covered by another category. |
|
||||||
|
|
||||||
|
The categories do not represent motives, relationships, state, or events that
|
||||||
|
the cited transcript does not establish. An `other` entry is not a substitute
|
||||||
|
for uncertain classification.
|
||||||
|
|
||||||
|
## Identity, evidence, and order
|
||||||
|
|
||||||
|
The required normalized [NPC artifact](dnd-npc-artifacts.md) resolves `name`.
|
||||||
|
Registry references are provenance only and never replace an interaction's own
|
||||||
|
evidence. Normalization canonicalizes recognized registry names, orders and
|
||||||
|
deduplicates exact source references, then orders interactions by valid source
|
||||||
|
chronology, NPC comparison identity, display name, kind, and reference sequence.
|
||||||
|
Only entries with the same canonical name, kind, and complete valid evidence
|
||||||
|
sequence are collapsed; distinct categories or evidence remain separate.
|
||||||
|
|
||||||
|
See the [combat-turn artifact](dnd-combat-turn-artifacts.md) for combat-action
|
||||||
|
occurrences and the [JSON output contract](json-output.md) for publication.
|
||||||
|
Pipeline mechanics are described in [D&D module internals](../internal/dnd.md).
|
||||||
69
docs/integrations/dnd-scene-description-artifacts.md
Normal file
69
docs/integrations/dnd-scene-description-artifacts.md
Normal file
@@ -0,0 +1,69 @@
|
|||||||
|
# D&D Scene-Description Artifact
|
||||||
|
|
||||||
|
This contract defines the durable output of `dnd/scene-descriptions`. Each
|
||||||
|
record classifies one accepted transcript chunk and gives it a minimal
|
||||||
|
source-grounded title and summary.
|
||||||
|
|
||||||
|
## Identity and compatibility
|
||||||
|
|
||||||
|
| Property | Value |
|
||||||
|
| --- | --- |
|
||||||
|
| Artifact kind | `dnd/scene-description-list` |
|
||||||
|
| Schema ID | `notarius.dnd.scene_descriptions` |
|
||||||
|
| Schema name | `notarius_dnd_scene_descriptions_v1` |
|
||||||
|
| Schema version | `v1` |
|
||||||
|
| Media type | `application/json` |
|
||||||
|
|
||||||
|
`v1` is a strict JSON object with required non-empty `scenes`. Scene and
|
||||||
|
source-reference objects reject unknown fields. An incompatible shape change
|
||||||
|
requires a new schema version.
|
||||||
|
|
||||||
|
## Wire shape
|
||||||
|
|
||||||
|
Each scene has exactly these required fields:
|
||||||
|
|
||||||
|
| Field | Contract |
|
||||||
|
| --- | --- |
|
||||||
|
| `id` | Non-empty accepted chunk ID, assigned by Notarius. |
|
||||||
|
| `source_ref` | The assigned inclusive source range for that chunk. |
|
||||||
|
| `kind` | `combat`, `narrative`, `recap`, or `meta`. |
|
||||||
|
| `title` | Non-empty, trimmed, source-grounded title. |
|
||||||
|
| `summary` | Non-empty, trimmed, source-grounded summary. |
|
||||||
|
|
||||||
|
`source_ref` has exactly `source_id`, `start_unit_id`, and `end_unit_id`.
|
||||||
|
Its source ID identifies the input transcript; its positive unit IDs identify
|
||||||
|
the chunk's inclusive range, with the start no later than the end.
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"scenes": [
|
||||||
|
{
|
||||||
|
"id": "chunk-000001",
|
||||||
|
"source_ref": {"source_id": "session-7", "start_unit_id": 1, "end_unit_id": 3},
|
||||||
|
"kind": "narrative",
|
||||||
|
"title": "Arrival at the watchtower",
|
||||||
|
"summary": "The party reaches the ruined watchtower and begins to investigate it."
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
## Meaning and normalized form
|
||||||
|
|
||||||
|
`combat` identifies a chunk where active combat is the central activity.
|
||||||
|
`narrative` is current in-world play that is not principally combat, recap, or
|
||||||
|
meta discussion. `recap` is primarily a recounting of an earlier session, and
|
||||||
|
`meta` is primarily out-of-character discussion. The artifact does not add
|
||||||
|
participants, confidence, events, or information absent from the chunk.
|
||||||
|
|
||||||
|
Normalization trims title and summary, orders scenes by source position and
|
||||||
|
then ID, and removes exact duplicate records. A reused ID with different
|
||||||
|
durable fields, or the same source range with different kind, title, or
|
||||||
|
summary, is invalid. It does not merge adjacent ranges, alter prose, or infer
|
||||||
|
missing scenes.
|
||||||
|
|
||||||
|
The [combat-turn artifact](dnd-combat-turn-artifacts.md) uses an exact matching
|
||||||
|
`combat` scene only as eligibility control; scene title, summary, and source
|
||||||
|
reference never become combat evidence. Publication is defined by the
|
||||||
|
[JSON output contract](json-output.md); implementation details live in
|
||||||
|
[D&D module internals](../internal/dnd.md).
|
||||||
@@ -1,152 +1,73 @@
|
|||||||
# D&D Spell-Cast Artifacts
|
# D&D Spell Artifact
|
||||||
|
|
||||||
This document is the durable artifact contract for approved
|
This contract defines the durable output of the `dnd/spells` extractor and
|
||||||
`dnd.spell_cast` artifacts produced by the implemented `dnd/spells` extractor.
|
normalizer. It records source-grounded spell-casting occurrences; it is not a
|
||||||
|
spellbook, a rules lookup result, or a record of hypothetical casts.
|
||||||
|
|
||||||
## Artifact Identity
|
## Identity and compatibility
|
||||||
|
|
||||||
- Extractor key: `dnd/spells`
|
| Property | Value |
|
||||||
- Artifact type: `dnd.spell_cast`
|
| --- | --- |
|
||||||
- Schema version: `v1`
|
| Artifact kind | `dnd/spell-list` |
|
||||||
- Prompt ID: `dnd.spells`
|
| Schema ID | `notarius.dnd.spells` |
|
||||||
- Response schema key: `dnd_spells`
|
| Schema name | `notarius_dnd_spells_v1` |
|
||||||
- Response schema ID: `notarius.dnd.spells`
|
| Schema version | `v1` |
|
||||||
- Response schema name: `notarius_dnd_spells_v1`
|
| Media type | `application/json` |
|
||||||
|
|
||||||
The extractor requires source chunks and transcript source capability. It
|
`v1` is a single strict JSON object. It requires `spell_casts`; the array may
|
||||||
returns generic artifact candidates that are serialized by the JSON output
|
be empty. Each spell-cast object and source-reference object rejects unknown
|
||||||
module.
|
fields. An incompatible shape change requires a new schema version.
|
||||||
|
|
||||||
## Artifact Envelope
|
## Wire shape
|
||||||
|
|
||||||
Approved artifacts use the generic artifact envelope documented in
|
Each `spell_casts` entry has these required fields:
|
||||||
[JSON Output](json-output.md#artifact-files):
|
|
||||||
|
|
||||||
```json
|
| Field | Contract |
|
||||||
{
|
| --- | --- |
|
||||||
"extractor_key": "dnd/spells",
|
| `caster` | Non-empty in-world character or creature name. |
|
||||||
"artifact_type": "dnd.spell_cast",
|
| `spell` | Non-empty spell name. |
|
||||||
"schema_version": "v1",
|
| `source_refs` | One or more transcript evidence ranges. |
|
||||||
"payload": {
|
|
||||||
"caster": "Aria",
|
|
||||||
"spell": "Cure Wounds",
|
|
||||||
"effect": "heals an injured ally",
|
|
||||||
"narrative_description": "Aria raises her holy symbol and casts Cure Wounds."
|
|
||||||
},
|
|
||||||
"source_refs": [
|
|
||||||
{
|
|
||||||
"source_id": "session-alpha",
|
|
||||||
"start_unit_id": "seg-001",
|
|
||||||
"end_unit_id": "seg-001"
|
|
||||||
}
|
|
||||||
]
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
## Payload Fields
|
Every source reference has exactly `source_id`, `start_unit_id`, and
|
||||||
|
`end_unit_id`. The source ID identifies the input transcript; the unit IDs are
|
||||||
The `payload` object contains:
|
positive inclusive unit identifiers, and the start may not follow the end in
|
||||||
|
that source. References are evidence for the cast, not campaign-reference or
|
||||||
- `caster`: in-world character or creature casting the spell;
|
NPC-registry provenance.
|
||||||
- `spell`: spell name;
|
|
||||||
- `effect`: concise spell effect in the scene;
|
|
||||||
- `narrative_description`: short description of the spell cast in context.
|
|
||||||
|
|
||||||
All payload fields are strings and must be non-empty after trimming.
|
|
||||||
|
|
||||||
`caster` is the in-world caster, not the transcript speaker.
|
|
||||||
|
|
||||||
## Source References
|
|
||||||
|
|
||||||
Source references live on the artifact envelope as `source_refs`; they are not
|
|
||||||
duplicated inside the `payload`.
|
|
||||||
|
|
||||||
Each source reference uses the generic source-reference shape:
|
|
||||||
|
|
||||||
- `source_id`
|
|
||||||
- `start_unit_id`
|
|
||||||
- `end_unit_id`
|
|
||||||
|
|
||||||
Validation requires:
|
|
||||||
|
|
||||||
- at least one source reference;
|
|
||||||
- non-empty source ID and unit IDs;
|
|
||||||
- source ID matching the source document ID;
|
|
||||||
- start and end unit IDs existing in the source document;
|
|
||||||
- start unit appearing before or at the same position as end unit.
|
|
||||||
|
|
||||||
## Structured LLM Response Shape
|
|
||||||
|
|
||||||
The extractor asks the LLM for this top-level response shape:
|
|
||||||
|
|
||||||
```json
|
```json
|
||||||
{
|
{
|
||||||
"spell_casts": [
|
"spell_casts": [
|
||||||
{
|
{
|
||||||
"caster": "Aria",
|
"caster": "Mira Thorn",
|
||||||
"spell": "Cure Wounds",
|
"spell": "Fireball",
|
||||||
"effect": "heals an injured ally",
|
|
||||||
"narrative_description": "Aria raises her holy symbol and casts Cure Wounds.",
|
|
||||||
"source_refs": [
|
"source_refs": [
|
||||||
{
|
{"source_id": "session-7", "start_unit_id": 12, "end_unit_id": 13}
|
||||||
"source_id": "session-alpha",
|
|
||||||
"start_unit_id": "seg-001",
|
|
||||||
"end_unit_id": "seg-001"
|
|
||||||
}
|
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
`spell_casts` must be present. It may be empty when no spell casts are found.
|
## Evidence and normalized form
|
||||||
|
|
||||||
The response schema asset is embedded at
|
An entry represents an actual cast or an unambiguous declared attempt. A spell
|
||||||
`internal/modules/extract/dnd/spells/assets/schemas/dnd_spells.v1.json`.
|
mention, rules discussion, plan, or catalog match alone is not an occurrence.
|
||||||
|
The configured catalog checks the name; it does not establish evidence.
|
||||||
|
|
||||||
## Validators
|
When normalization is selected, recognized spell names use the effective
|
||||||
|
catalog's canonical display name. Source references are put in canonical source
|
||||||
|
order and exact duplicate references are removed. A later entry is collapsed
|
||||||
|
only when it has the same canonical spell, the same case- and
|
||||||
|
whitespace-insensitive caster identity, and the same complete valid reference
|
||||||
|
sequence. Remaining entries retain their merged order.
|
||||||
|
|
||||||
The extractor supplies two deterministic validators by default:
|
The optional normalized [NPC artifact](dnd-npc-artifacts.md) can ground a
|
||||||
|
caster name. Its own references remain registry provenance and are never copied
|
||||||
|
into `source_refs`.
|
||||||
|
|
||||||
- `dnd/spells/shape`
|
## Related contracts
|
||||||
- `dnd/spells/source_refs`
|
|
||||||
|
|
||||||
Rejection reason codes:
|
The [spell-catalog overlay contract](dnd-spell-catalog-overlays.md) defines
|
||||||
|
the configured catalog additions. The [JSON output contract](json-output.md)
|
||||||
- `invalid_payload`: payload JSON cannot be decoded as a spell-cast payload.
|
defines where this logical artifact is published; [D&D module internals](../internal/dnd.md)
|
||||||
- `missing_required_field`: `caster`, `spell`, `effect`, or
|
describes extraction and validation mechanics.
|
||||||
`narrative_description` is blank.
|
|
||||||
- `missing_source_ref`: candidate has no source references.
|
|
||||||
- `invalid_source_ref`: at least one source reference fails generic source
|
|
||||||
reference validation.
|
|
||||||
|
|
||||||
Rejected candidates are written to `rejected.json` by the JSON output module.
|
|
||||||
|
|
||||||
## Manifest Metadata
|
|
||||||
|
|
||||||
The extractor adds prompt and response-schema provenance under the artifact lane
|
|
||||||
manifest metadata:
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"metadata": {
|
|
||||||
"extractor": {
|
|
||||||
"prompt_id": "dnd.spells",
|
|
||||||
"prompt_version": "v1",
|
|
||||||
"prompt_sha256": "sha256:...",
|
|
||||||
"response_schema_key": "dnd_spells",
|
|
||||||
"response_schema_id": "notarius.dnd.spells",
|
|
||||||
"response_schema_name": "notarius_dnd_spells_v1",
|
|
||||||
"response_schema_version": "v1",
|
|
||||||
"response_schema_sha256": "sha256:..."
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
Raw prompt and schema content are not included in manifest metadata.
|
|
||||||
|
|
||||||
## Compatibility Limit
|
|
||||||
|
|
||||||
This contract covers only `dnd.spell_cast` artifacts produced by the
|
|
||||||
implemented spell-cast extractor.
|
|
||||||
|
|||||||
72
docs/integrations/dnd-spell-catalog-overlays.md
Normal file
72
docs/integrations/dnd-spell-catalog-overlays.md
Normal file
@@ -0,0 +1,72 @@
|
|||||||
|
# D&D Spell-Catalog Overlays
|
||||||
|
|
||||||
|
This document defines the optional JSON overlay consumed by the D&D spell
|
||||||
|
extractor. An overlay contributes campaign spell names and aliases for
|
||||||
|
recognition. It does not define spell rules, effects, levels, classes, or
|
||||||
|
transcript evidence. Bind the optional `spell_catalog` reference as described
|
||||||
|
in [Configuration](../config.md#references-and-ordered-handoffs).
|
||||||
|
|
||||||
|
## Contract Identity
|
||||||
|
|
||||||
|
| Property | Value |
|
||||||
|
| --- | --- |
|
||||||
|
| Consumer | D&D spell extraction and normalization |
|
||||||
|
| Reference slot | `spell_catalog` |
|
||||||
|
| Media type | `application/json` |
|
||||||
|
| Required schema version | `notarius.dnd.spell-catalog-overlay.v1` |
|
||||||
|
| Base catalog | Embedded D&D 5e 2014 SRD catalog |
|
||||||
|
|
||||||
|
At most one overlay document may be bound. The maintained example is
|
||||||
|
[dnd-spell-catalog.json](../../examples/dnd-spell-catalog.json).
|
||||||
|
|
||||||
|
## Wire Shape
|
||||||
|
|
||||||
|
This is a minimal valid overlay:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"schema_version": "notarius.dnd.spell-catalog-overlay.v1",
|
||||||
|
"catalogs": [
|
||||||
|
{
|
||||||
|
"id": "campaign.example",
|
||||||
|
"ruleset": "dnd-5e-2014",
|
||||||
|
"source": {"title": "Example campaign spells"},
|
||||||
|
"spells": [{"name": "Aegis of Emberfall"}]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
| Field | Required | Meaning and constraints |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| `schema_version` | Yes | Exactly `notarius.dnd.spell-catalog-overlay.v1`. |
|
||||||
|
| `catalogs` | Yes | Non-empty array of catalog objects with unique IDs. |
|
||||||
|
| `catalogs[].id` | Yes | Non-empty trimmed string. |
|
||||||
|
| `catalogs[].ruleset` | Yes | Exactly `dnd-5e-2014`. |
|
||||||
|
| `catalogs[].source.title` | Yes | Non-empty trimmed string. |
|
||||||
|
| `catalogs[].source.version` | No | String when present. |
|
||||||
|
| `catalogs[].source.url` | No | String when present. |
|
||||||
|
| `catalogs[].source.license` | No | String when present. |
|
||||||
|
| `catalogs[].spells` | Yes | Non-empty array of spell objects. |
|
||||||
|
| `catalogs[].spells[].name` | Yes | Non-empty trimmed string. |
|
||||||
|
| `catalogs[].spells[].aliases` | No | Array of non-empty trimmed strings when present. |
|
||||||
|
|
||||||
|
Unknown fields are rejected at every object level. The document must contain
|
||||||
|
one JSON value; `null` is not accepted for optional strings or aliases.
|
||||||
|
|
||||||
|
## Composition And Compatibility
|
||||||
|
|
||||||
|
Notarius starts with the embedded base catalog, then applies overlay catalogs
|
||||||
|
in ascending catalog-ID order. A new canonical spell name adds a recognition
|
||||||
|
entry. If an overlay names an existing canonical spell, it augments that spell
|
||||||
|
with aliases while retaining the established display spelling.
|
||||||
|
|
||||||
|
Repeated aliases for the same spell are accepted. A canonical-name, canonical-
|
||||||
|
to-alias, or alias-to-alias collision between different spells is rejected,
|
||||||
|
including a collision with the embedded catalog. Matching uses the catalog’s
|
||||||
|
case, whitespace, and apostrophe normalization, so authors should avoid names
|
||||||
|
or aliases that normalize to another spell.
|
||||||
|
|
||||||
|
The overlay is a recognition aid only. The durable spell-artifact schema and
|
||||||
|
source-evidence rules are defined by the
|
||||||
|
[D&D spell artifact contract](dnd-spell-artifacts.md).
|
||||||
116
docs/integrations/evidence-context.md
Normal file
116
docs/integrations/evidence-context.md
Normal file
@@ -0,0 +1,116 @@
|
|||||||
|
# Published Evidence Context
|
||||||
|
|
||||||
|
This contract defines the optional `source/evidence-context` artifact emitted
|
||||||
|
by the production JSON output. Its configuration is owned by
|
||||||
|
[Configuration](../config.md#module-bindings-and-validators); its logical-file
|
||||||
|
discovery is owned by [Published JSON Output](json-output.md).
|
||||||
|
|
||||||
|
## Identity And Discovery
|
||||||
|
|
||||||
|
When enabled, the JSON bundle contains `evidence-context.json` and an
|
||||||
|
`index.json` `evidence_context` descriptor with the same six fields as other
|
||||||
|
pipeline-wide artifact descriptors.
|
||||||
|
|
||||||
|
| Property | Value |
|
||||||
|
| --- | --- |
|
||||||
|
| Artifact kind | `source/evidence-context` |
|
||||||
|
| Media type | `application/json` |
|
||||||
|
| Schema ID | `notarius.source.evidence_context` |
|
||||||
|
| Schema name | `notarius_source_evidence_context_v1` |
|
||||||
|
| Schema version | `v1` |
|
||||||
|
| Logical file | `evidence-context.json` |
|
||||||
|
|
||||||
|
Consumers must discover the file from the descriptor, verify all six descriptor
|
||||||
|
fields, and decode only a supported schema version. The descriptor is optional:
|
||||||
|
its absence means evidence publication was not enabled for that bundle.
|
||||||
|
|
||||||
|
## Payload
|
||||||
|
|
||||||
|
The v1 payload is a JSON object with required `source_id`, `source_digest`,
|
||||||
|
`window_units`, `selected_lanes`, and `contexts` fields. `selected_lanes` and
|
||||||
|
`contexts` are always arrays; an enabled configuration with no accepted direct
|
||||||
|
evidence publishes `contexts: []`.
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"source_id": "session-alpha",
|
||||||
|
"source_digest": "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef",
|
||||||
|
"window_units": 1,
|
||||||
|
"selected_lanes": ["npcs", "spells"],
|
||||||
|
"contexts": [
|
||||||
|
{
|
||||||
|
"context_ref": {
|
||||||
|
"source_id": "session-alpha",
|
||||||
|
"start_unit_id": 10,
|
||||||
|
"end_unit_id": 20
|
||||||
|
},
|
||||||
|
"evidence_refs": [
|
||||||
|
{
|
||||||
|
"lane_id": "spells",
|
||||||
|
"source_ref": {
|
||||||
|
"source_id": "session-alpha",
|
||||||
|
"start_unit_id": 10,
|
||||||
|
"end_unit_id": 10
|
||||||
|
}
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"units": [
|
||||||
|
{
|
||||||
|
"id": 10,
|
||||||
|
"kind": "transcript_segment",
|
||||||
|
"text": "Aria casts Cure Wounds.",
|
||||||
|
"ref": {
|
||||||
|
"source_id": "session-alpha",
|
||||||
|
"start_unit_id": 10,
|
||||||
|
"end_unit_id": 10
|
||||||
|
}
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 20,
|
||||||
|
"kind": "transcript_segment",
|
||||||
|
"text": "The party regroups.",
|
||||||
|
"ref": {
|
||||||
|
"source_id": "session-alpha",
|
||||||
|
"start_unit_id": 20,
|
||||||
|
"end_unit_id": 20
|
||||||
|
}
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Each context requires `context_ref`, `evidence_refs`, and `units` arrays.
|
||||||
|
`context_ref` identifies the first and last included unit. Each evidence entry
|
||||||
|
contains a selected `lane_id` and an original `source_ref`. A unit uses the
|
||||||
|
existing source-unit shape: required `id`, `kind`, `text`, and self `ref`, plus
|
||||||
|
optional JSON-object `metadata`. Fixed payload objects reject unknown fields;
|
||||||
|
unit metadata may contain application-defined JSON values.
|
||||||
|
|
||||||
|
## Citations And Context
|
||||||
|
|
||||||
|
`evidence_refs` are the authoritative citations. They identify the direct
|
||||||
|
references emitted by accepted normalized artifacts. `context_ref` and the
|
||||||
|
units collection include those cited units plus nearby source units selected by
|
||||||
|
the configured window. They are explanatory context, not widened citations.
|
||||||
|
|
||||||
|
Only accepted outputs from the configured lane allowlist contribute. Rejected,
|
||||||
|
failed, absent, and lane-filtered outputs do not contribute. The artifact never
|
||||||
|
contains raw input bytes, prompts, model responses, auxiliary reference
|
||||||
|
content, credentials, or filesystem paths.
|
||||||
|
|
||||||
|
## Ordering And Compatibility
|
||||||
|
|
||||||
|
The selected lane allowlist is lexical. Contexts and units are in source
|
||||||
|
document position order, not numeric unit-ID order. Direct evidence entries
|
||||||
|
are deterministically ordered by lane and source reference. Overlapping or
|
||||||
|
contiguous windows merge, and each source unit appears at most once in the
|
||||||
|
resulting contexts.
|
||||||
|
|
||||||
|
The artifact is additive to the JSON bundle and is not a lane payload,
|
||||||
|
normalized-output count, checkpoint, or generated reference. Consumers that
|
||||||
|
do not need it must tolerate the absent optional descriptor. Consumers that do
|
||||||
|
use it should preserve the artifact and its schema identity with the run
|
||||||
|
provenance, and should treat its source text and metadata as sensitive durable
|
||||||
|
content.
|
||||||
@@ -1,192 +1,118 @@
|
|||||||
# JSON Output
|
# Published JSON Output
|
||||||
|
|
||||||
This document is the durable JSON output file-format contract produced by the
|
This document defines the logical JSON bundle emitted by the production JSON
|
||||||
implemented `json` output module and written by the CLI.
|
output encoder. The bundle’s physical destination, atomic publication, and
|
||||||
|
retention are operational concerns; see [Operations](../operations.md#output-bundles).
|
||||||
|
Output configuration, including chunk-map and evidence-context publication, belongs in
|
||||||
|
[Configuration](../config.md#module-bindings-and-validators).
|
||||||
|
|
||||||
## Output Directory
|
## Bundle Layout
|
||||||
|
|
||||||
The CLI writes logical output files under:
|
All paths below are logical, relative, slash-separated bundle paths. The
|
||||||
|
encoder always emits the first four JSON files below and adds lane or
|
||||||
|
pipeline-wide artifact files when their corresponding artifacts are available:
|
||||||
|
|
||||||
```text
|
A subprocess caller first obtains the physical bundle root from the
|
||||||
<output-root>/<run-id>/
|
[run-result receipt](run-result.md), then resolves `index.json` beneath that
|
||||||
```
|
root for the logical discovery described here.
|
||||||
|
|
||||||
The default output root is `./notarius-output`. Operational behavior is covered
|
| Path | Purpose |
|
||||||
in [Operations](../operations.md).
|
| --- | --- |
|
||||||
|
| `index.json` | Entry point that names the other published files and lane payloads. |
|
||||||
|
| `manifest.json` | Run provenance and result summaries. |
|
||||||
|
| `rejected.json` | Rejected pipeline outputs. |
|
||||||
|
| `warnings.json` | Accepted-output and run warnings. |
|
||||||
|
| `lanes/<safe-lane-id>.json` | One normalized artifact payload for each lane. |
|
||||||
|
| `chunk-map.json` | Optional accepted chunk map, when its export is enabled and available. |
|
||||||
|
| `evidence-context.json` | Optional source-context artifact, when evidence publication is enabled. |
|
||||||
|
|
||||||
## Files
|
JSON files are pretty-printed with a trailing newline. Lane payloads are
|
||||||
|
accepted only when their media type is `application/json`.
|
||||||
The `json` output module writes:
|
|
||||||
|
|
||||||
- `index.json`
|
|
||||||
- `manifest.json`
|
|
||||||
- `artifacts/<artifact-type>.json`, one file per approved artifact type
|
|
||||||
- `rejected.json`
|
|
||||||
- `warnings.json`
|
|
||||||
|
|
||||||
Files are pretty-printed JSON with a trailing newline.
|
|
||||||
|
|
||||||
## `index.json`
|
## `index.json`
|
||||||
|
|
||||||
Shape:
|
`index.json` is the bundle’s discovery document. An approved run with no
|
||||||
|
normalized lanes has this valid minimal index:
|
||||||
|
|
||||||
```json
|
```json
|
||||||
{
|
{
|
||||||
"manifest_file": "manifest.json",
|
"manifest_file": "manifest.json",
|
||||||
"artifact_files": [
|
"output_files": [],
|
||||||
{
|
|
||||||
"artifact_type": "dnd.spell_cast",
|
|
||||||
"file": "artifacts/dnd.spell_cast.json"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"rejected_file": "rejected.json",
|
"rejected_file": "rejected.json",
|
||||||
"warnings_file": "warnings.json"
|
"warnings_file": "warnings.json"
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
`artifact_files` is sorted by artifact type. It is empty when no artifacts are
|
| Field | Required | Meaning |
|
||||||
approved.
|
| --- | --- | --- |
|
||||||
|
| `manifest_file` | Yes | Always `manifest.json`. |
|
||||||
|
| `output_files` | Yes | Lane descriptors sorted by `lane_id`. |
|
||||||
|
| `rejected_file` | Yes | Always `rejected.json`. |
|
||||||
|
| `warnings_file` | Yes | Always `warnings.json`. |
|
||||||
|
| `chunk_map` | No | Descriptor for the pipeline-wide `chunk-map.json`; never a lane descriptor. |
|
||||||
|
| `evidence_context` | No | Descriptor for the pipeline-wide `evidence-context.json`; never a lane descriptor. |
|
||||||
|
|
||||||
|
Each lane descriptor has required `lane_id` and `file`. It may also include
|
||||||
|
`media_type`, `module_key`, `schema_id`, `schema_name`, and `schema_version`
|
||||||
|
when supplied by the normalized artifact. Each pipeline-wide artifact
|
||||||
|
descriptor (`chunk_map` or `evidence_context`) contains `artifact_kind`,
|
||||||
|
`file`, `media_type`, `schema_id`, `schema_name`, and `schema_version`. Their
|
||||||
|
payloads are defined by the [Accepted Chunk Map contract](chunk-map.md) and
|
||||||
|
[Published Evidence Context](evidence-context.md), respectively.
|
||||||
|
|
||||||
|
The lane path is derived from its lane ID. Characters outside letters, digits,
|
||||||
|
periods, underscores, and hyphens become underscores; `..` sequences are
|
||||||
|
neutralized; leading and trailing periods and underscores are removed. A lane
|
||||||
|
that produces an empty name, or two lanes that produce the same path, makes
|
||||||
|
output encoding fail.
|
||||||
|
|
||||||
|
## Lane Payloads
|
||||||
|
|
||||||
|
Each `lanes/<safe-lane-id>.json` file is the codec-owned normalized JSON for
|
||||||
|
that lane. Consumers should use the index descriptor’s schema identity rather
|
||||||
|
than infer a lane schema from its name. The current D&D payload contracts are
|
||||||
|
[spells](dnd-spell-artifacts.md), [NPCs](dnd-npc-artifacts.md),
|
||||||
|
[NPC interactions](dnd-npc-interaction-artifacts.md),
|
||||||
|
[combat turns](dnd-combat-turn-artifacts.md),
|
||||||
|
[item events](dnd-item-event-artifacts.md), and
|
||||||
|
[scene descriptions](dnd-scene-description-artifacts.md).
|
||||||
|
|
||||||
## `manifest.json`
|
## `manifest.json`
|
||||||
|
|
||||||
`manifest.json` contains a run manifest:
|
`manifest.json` is published provenance, not a copy of lane payloads or a
|
||||||
|
checkpoint store. Fields without a value may be omitted. Its top-level fields
|
||||||
|
group into the following externally observable summaries:
|
||||||
|
|
||||||
```json
|
| Group | Fields |
|
||||||
{
|
| --- | --- |
|
||||||
"run_id": "run-123",
|
| Run identity and result | `run_id`, `pipeline_id`, `pipeline_digest`, `schema_version`, `validation_status`, `started_at`, `completed_at` |
|
||||||
"pipeline_id": "dnd-session",
|
| Resolved components | `input_module`, `chunker`, `extractors`, `merger`, `normalizer`, `output_encoder`, `artifact_lanes`, `validator_chains`, `module_metadata` |
|
||||||
"pipeline_digest": "sha256:...",
|
| Source and references | `source_digests`, `references` |
|
||||||
"input_module": "seriatim",
|
| Published result summaries | `normalized_outputs`, `rejected_outputs` |
|
||||||
"chunker": "generic",
|
| Execution summaries | `chunk_plan`, `checkpoint_decisions`, `llm_profiles`, `metadata` |
|
||||||
"source_digests": ["sha256:..."],
|
|
||||||
"extractors": ["dnd/spells"],
|
|
||||||
"merger": "appendorder",
|
|
||||||
"normalizer": "noop",
|
|
||||||
"output_encoder": "json",
|
|
||||||
"artifact_lanes": [
|
|
||||||
{
|
|
||||||
"id": "spells",
|
|
||||||
"extractor": "dnd/spells",
|
|
||||||
"merger": "appendorder",
|
|
||||||
"normalizer": "noop"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"llm_profiles": [
|
|
||||||
{
|
|
||||||
"id": "default",
|
|
||||||
"provider": "openai-compatible",
|
|
||||||
"model": "configured-model"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"validation_status": "approved",
|
|
||||||
"started_at": "2026-01-01T00:00:00Z",
|
|
||||||
"completed_at": "2026-01-01T00:00:01Z"
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
Fields with empty values may be omitted by JSON encoding.
|
`references` records provenance such as the target, slot, origin, digest,
|
||||||
|
media type, size, and generated-artifact identity. It does not contain
|
||||||
|
reference content. `normalized_outputs` and `rejected_outputs` likewise
|
||||||
|
summarize results without embedding lane payload bytes. A chunk-plan summary is
|
||||||
|
provenance for the plan used by this run; cache records, debug artifacts, and
|
||||||
|
other operational state are not published as bundle files.
|
||||||
|
|
||||||
`validation_status` is `approved` when no candidates were rejected and
|
## Rejections And Warnings
|
||||||
`rejected` when one or more candidates were rejected.
|
|
||||||
|
|
||||||
## Artifact Files
|
`rejected.json` is always an object with a `rejected` array. Each entry has
|
||||||
|
required `stage` and `message`; `step_id`, `lane_id`, `module_key`, `chunk_id`,
|
||||||
|
`chunk_index`, `validator_name`, `reason_code`, `attempt_count`, and
|
||||||
|
`diagnostic_artifact_path` are present only when applicable.
|
||||||
|
|
||||||
Each artifact file has this shape:
|
`warnings.json` is always an object with a `warnings` array. Each warning has
|
||||||
|
`reason_code` and `message`; `scope` is optional. Both arrays are empty when
|
||||||
|
there is nothing to report.
|
||||||
|
|
||||||
```json
|
## Compatibility
|
||||||
{
|
|
||||||
"artifact_type": "dnd.spell_cast",
|
|
||||||
"artifacts": [
|
|
||||||
{
|
|
||||||
"extractor_key": "dnd/spells",
|
|
||||||
"artifact_type": "dnd.spell_cast",
|
|
||||||
"schema_version": "v1",
|
|
||||||
"payload": {},
|
|
||||||
"source_refs": [
|
|
||||||
{
|
|
||||||
"source_id": "session-alpha",
|
|
||||||
"start_unit_id": "seg-001",
|
|
||||||
"end_unit_id": "seg-001"
|
|
||||||
}
|
|
||||||
]
|
|
||||||
}
|
|
||||||
]
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
Artifact envelope fields:
|
The index is the authoritative map from a logical lane to its published
|
||||||
|
payload. Consumers must tolerate omitted optional manifest and descriptor
|
||||||
- `extractor_key`: extractor module key.
|
fields, and should rely on the linked artifact contract for each lane’s JSON
|
||||||
- `artifact_type`: artifact type.
|
shape. This contract describes the published logical bundle only; it does not
|
||||||
- `schema_version`: artifact schema version.
|
promise a filesystem layout or expose internal state formats.
|
||||||
- `payload`: artifact-type-specific JSON payload.
|
|
||||||
- `source_refs`: optional generic source references.
|
|
||||||
- `metadata`: optional artifact metadata.
|
|
||||||
|
|
||||||
Artifact file names are produced by sanitizing the artifact type:
|
|
||||||
|
|
||||||
- characters outside `A-Z`, `a-z`, `0-9`, `.`, `_`, and `-` become `_`;
|
|
||||||
- repeated `..` sequences are replaced;
|
|
||||||
- leading and trailing `.`, `_`, and `-` are trimmed;
|
|
||||||
- empty sanitized names are rejected.
|
|
||||||
|
|
||||||
For current D&D spell-cast artifacts, the file is
|
|
||||||
`artifacts/dnd.spell_cast.json`.
|
|
||||||
|
|
||||||
## `rejected.json`
|
|
||||||
|
|
||||||
Shape:
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"rejected": [
|
|
||||||
{
|
|
||||||
"candidate": {
|
|
||||||
"index": 0,
|
|
||||||
"extractor_key": "dnd/spells",
|
|
||||||
"artifact_type": "dnd.spell_cast",
|
|
||||||
"schema_version": "v1",
|
|
||||||
"payload": {},
|
|
||||||
"source_refs": []
|
|
||||||
},
|
|
||||||
"validator_name": "dnd/spells/source_refs",
|
|
||||||
"reason_code": "missing_source_ref",
|
|
||||||
"message": "spell cast candidate must include at least one source ref"
|
|
||||||
}
|
|
||||||
]
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
`rejected` is an empty array when no candidates are rejected.
|
|
||||||
|
|
||||||
## `warnings.json`
|
|
||||||
|
|
||||||
Shape:
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"warnings": [
|
|
||||||
{
|
|
||||||
"scope": "output",
|
|
||||||
"reason_code": "example_warning",
|
|
||||||
"message": "warning message"
|
|
||||||
}
|
|
||||||
]
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
`warnings` is an empty array when no warnings are reported.
|
|
||||||
|
|
||||||
## Path Safety
|
|
||||||
|
|
||||||
The output module returns slash-separated logical paths. The CLI also validates
|
|
||||||
logical output names before writing:
|
|
||||||
|
|
||||||
- names must be non-empty;
|
|
||||||
- names must be relative;
|
|
||||||
- names must be clean;
|
|
||||||
- names must use `/`, not `\`;
|
|
||||||
- names must not contain `..`;
|
|
||||||
- resolved paths must stay under the run output directory.
|
|
||||||
|
|
||||||
Durable writes are atomic per file.
|
|
||||||
|
|||||||
@@ -1,128 +0,0 @@
|
|||||||
# OpenAI-Compatible Structured Output
|
|
||||||
|
|
||||||
This document describes the external LLM provider contract implemented by the
|
|
||||||
production Notarius LLM client.
|
|
||||||
|
|
||||||
## Provider
|
|
||||||
|
|
||||||
- Provider key: `openai-compatible`
|
|
||||||
- HTTP method: `POST`
|
|
||||||
- Endpoint: `<base_url>/chat/completions`
|
|
||||||
- Request body: JSON
|
|
||||||
- Response mode: chat completions with structured JSON schema output
|
|
||||||
|
|
||||||
`base_url` is trimmed of trailing slashes before `/chat/completions` is
|
|
||||||
appended. Configure provider settings in [Configuration](../config.md).
|
|
||||||
|
|
||||||
## Request
|
|
||||||
|
|
||||||
The client sends a JSON object with:
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"model": "configured-model",
|
|
||||||
"messages": [
|
|
||||||
{
|
|
||||||
"role": "system",
|
|
||||||
"content": "..."
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"role": "user",
|
|
||||||
"content": "..."
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"response_format": {
|
|
||||||
"type": "json_schema",
|
|
||||||
"json_schema": {
|
|
||||||
"name": "schema_name",
|
|
||||||
"strict": true,
|
|
||||||
"schema": {}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
Implemented request behavior:
|
|
||||||
|
|
||||||
- `model` comes from the structured completion request when set, otherwise from
|
|
||||||
the configured LLM profile.
|
|
||||||
- `messages` must be non-empty; each role and content must be non-empty after
|
|
||||||
trimming.
|
|
||||||
- `response_format.type` is always `json_schema`.
|
|
||||||
- `response_format.json_schema.strict` is always `true`.
|
|
||||||
- `response_format.json_schema.name` and `schema` come from the extractor or
|
|
||||||
validator making the call.
|
|
||||||
|
|
||||||
If an API key is configured, the client sends:
|
|
||||||
|
|
||||||
```text
|
|
||||||
Authorization: Bearer <api-key>
|
|
||||||
```
|
|
||||||
|
|
||||||
The client always sends `Content-Type: application/json`.
|
|
||||||
|
|
||||||
## Response
|
|
||||||
|
|
||||||
The client expects a JSON response with at least one choice:
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"model": "provider-model",
|
|
||||||
"choices": [
|
|
||||||
{
|
|
||||||
"message": {
|
|
||||||
"content": "{\"field\":\"value\"}"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"usage": {
|
|
||||||
"prompt_tokens": 10,
|
|
||||||
"completion_tokens": 5,
|
|
||||||
"total_tokens": 15
|
|
||||||
}
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
`choices[0].message.content` may be either:
|
|
||||||
|
|
||||||
- a JSON string whose contents are valid JSON; or
|
|
||||||
- raw JSON.
|
|
||||||
|
|
||||||
The decoded content is unmarshaled into the caller-provided structured output
|
|
||||||
target. If `usage` is present, prompt, completion, and total token counts are
|
|
||||||
copied into the completion response.
|
|
||||||
|
|
||||||
## Errors And Retries
|
|
||||||
|
|
||||||
The client validates base URL, model, response schema name, response schema
|
|
||||||
JSON, messages, and output target before or during the call.
|
|
||||||
|
|
||||||
Retryable failures:
|
|
||||||
|
|
||||||
- HTTP request failure;
|
|
||||||
- response body read failure;
|
|
||||||
- HTTP `429`;
|
|
||||||
- HTTP `5xx`;
|
|
||||||
- malformed provider response envelope;
|
|
||||||
- missing choices;
|
|
||||||
- missing, empty, or invalid assistant JSON content;
|
|
||||||
- structured-output decode failure.
|
|
||||||
|
|
||||||
Non-retryable provider status codes include non-`429` `4xx` responses.
|
|
||||||
|
|
||||||
Provider error bodies are parsed for `error.message` or `message` when present.
|
|
||||||
Configured API key values and bearer-token values are redacted from returned
|
|
||||||
provider errors.
|
|
||||||
|
|
||||||
## Timeouts And Concurrency
|
|
||||||
|
|
||||||
The configured profile timeout is applied per provider request when greater
|
|
||||||
than zero. Context cancellation is respected.
|
|
||||||
|
|
||||||
The production CLI wraps the provider client with the LLM scheduler. Effective
|
|
||||||
concurrency is described in [LLM runtime internals](../internal/llm.md).
|
|
||||||
|
|
||||||
## Limits
|
|
||||||
|
|
||||||
This contract documents only the fields the implemented client sends and reads.
|
|
||||||
Provider-specific extensions are ignored unless they affect those fields.
|
|
||||||
68
docs/integrations/run-result.md
Normal file
68
docs/integrations/run-result.md
Normal file
@@ -0,0 +1,68 @@
|
|||||||
|
# Run Result Receipt
|
||||||
|
|
||||||
|
`notarius run --json` writes this receipt to standard output when a run
|
||||||
|
completes successfully. It lets a subprocess caller discover the physical root
|
||||||
|
of the published output bundle without parsing interactive command output.
|
||||||
|
Command syntax, streams, and exit statuses are defined in the
|
||||||
|
[CLI reference](../cli.md); logical files within the bundle are defined in the
|
||||||
|
[Published JSON Output contract](json-output.md).
|
||||||
|
|
||||||
|
## Schema
|
||||||
|
|
||||||
|
The current schema version is `notarius.run-result.v1`.
|
||||||
|
|
||||||
|
| Field | Required | Meaning |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| `schema_version` | Yes | Exactly `notarius.run-result.v1`. |
|
||||||
|
| `run_id` | Yes | The finalized Notarius run identifier. |
|
||||||
|
| `pipeline_id` | Yes | The effective pipeline identifier. |
|
||||||
|
| `output_directory` | Yes | Absolute path to the published, run-specific output bundle. |
|
||||||
|
| `index_file` | For the production JSON output | Logical path `index.json`; omitted for other output modules. |
|
||||||
|
| `normalized_output_count` | Yes | Number of final normalized outputs. |
|
||||||
|
| `rejected_output_count` | Yes | Number of recorded rejected outputs. |
|
||||||
|
| `warning_count` | Yes | Number of final run warnings. |
|
||||||
|
| `validation_status` | Yes | The final run manifest validation status. |
|
||||||
|
| `debug_directory` | No | Absolute path to the run-specific debug bundle when requested debug capture completed. |
|
||||||
|
|
||||||
|
For the production `json` output module, `index_file` is present only when the
|
||||||
|
completed run returned exactly one logical output file named `index.json`.
|
||||||
|
For another output module, its absence does not indicate a failed run.
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"schema_version": "notarius.run-result.v1",
|
||||||
|
"run_id": "run-1770000000000000000-0123456789abcdef0123456789abcdef",
|
||||||
|
"pipeline_id": "dnd-session",
|
||||||
|
"output_directory": "/work/results/run-1770000000000000000-0123456789abcdef0123456789abcdef",
|
||||||
|
"index_file": "index.json",
|
||||||
|
"normalized_output_count": 6,
|
||||||
|
"rejected_output_count": 2,
|
||||||
|
"warning_count": 1,
|
||||||
|
"validation_status": "rejected"
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
## Paths And Bundle Discovery
|
||||||
|
|
||||||
|
`output_directory` and `debug_directory`, when present, are lexical absolute
|
||||||
|
paths. They identify the paths used by Notarius and do not resolve symlinks.
|
||||||
|
`output_directory` is the run-specific bundle, not the configured output root.
|
||||||
|
|
||||||
|
The receipt is a summary and discovery document. It does not contain lane
|
||||||
|
descriptors, payloads, manifest data, rejections, warnings, or file contents.
|
||||||
|
For the production JSON output, resolve `index_file` beneath
|
||||||
|
`output_directory`, reject path escapes, and use the
|
||||||
|
[Published JSON Output contract](json-output.md) to discover logical files and
|
||||||
|
lane payloads.
|
||||||
|
|
||||||
|
## Delivery And Compatibility
|
||||||
|
|
||||||
|
Notarius writes the receipt only after the output bundle has been published and
|
||||||
|
any requested debug terminal reporting has completed. Standard output is not
|
||||||
|
transactional: a result-write failure returns a nonzero status and can leave
|
||||||
|
partial bytes. Consumers must ignore standard output unless the process exits
|
||||||
|
with status 0.
|
||||||
|
|
||||||
|
Future versions may add optional fields to this schema. Consumers must tolerate
|
||||||
|
unknown fields. An incompatible field or semantic change requires a new
|
||||||
|
`schema_version` value.
|
||||||
@@ -1,120 +1,73 @@
|
|||||||
# Seriatim Transcript JSON
|
# Seriatim Transcript Input
|
||||||
|
|
||||||
This document is the external input contract for the implemented `seriatim`
|
This document defines the JSON transcript accepted by the production Seriatim
|
||||||
input adapter.
|
input adapter. It is a source input, not a durable lane artifact. Configure the
|
||||||
|
input adapter through [Configuration](../config.md#production-module-keys).
|
||||||
|
|
||||||
## Adapter
|
## Contract Identity
|
||||||
|
|
||||||
- Module key: `seriatim`
|
| Property | Value |
|
||||||
- Document kind: `transcript`
|
| --- | --- |
|
||||||
- Unit kind: `transcript_segment`
|
| Consumer | Seriatim input adapter |
|
||||||
- Source format: `application/vnd.seriatim+json`
|
| Media type | `application/vnd.seriatim+json` |
|
||||||
|
| Source document kind | `transcript` |
|
||||||
The adapter parses raw Seriatim JSON into a generic source document. It owns
|
| Source-unit kind | `transcript_segment` |
|
||||||
transcript-specific JSON parsing and metadata mapping; core source and pipeline
|
|
||||||
code stay source-format agnostic.
|
|
||||||
|
|
||||||
## Accepted Shape
|
## Accepted Shape
|
||||||
|
|
||||||
The input must be one JSON object with top-level `metadata` and `segments`
|
The input is one JSON object containing `metadata` and a non-empty `segments`
|
||||||
fields. This covers the maintained minimal fixture and Seriatim intermediate
|
array. This minimal document is valid:
|
||||||
output that provides the same required segment fields.
|
|
||||||
|
|
||||||
```json
|
```json
|
||||||
{
|
{
|
||||||
"metadata": {
|
"metadata": {"id": "session-alpha"},
|
||||||
"id": "session-alpha",
|
|
||||||
"title": "Synthetic D&D spell session"
|
|
||||||
},
|
|
||||||
"segments": [
|
"segments": [
|
||||||
{
|
{
|
||||||
"id": "seg-001",
|
"id": 1,
|
||||||
"start": 0,
|
"start": 0,
|
||||||
"end": 4,
|
"end": 4,
|
||||||
"speaker": "Aria",
|
"speaker": "Aria",
|
||||||
"text": "Aria raises her holy symbol and casts Cure Wounds."
|
"text": "Aria casts Cure Wounds."
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
The maintained example is
|
The maintained two-segment input is
|
||||||
[examples/seriatim-minimal-transcript.json](../../examples/seriatim-minimal-transcript.json).
|
[seriatim-minimal-transcript.json](../../examples/seriatim-minimal-transcript.json).
|
||||||
|
|
||||||
Top-level metadata entries are preserved. Other segment fields, such as
|
| Field | Required | Meaning and constraints |
|
||||||
`categories`, are ignored.
|
| --- | --- | --- |
|
||||||
|
| `metadata` | Yes | JSON object. Its entries become source metadata; no particular metadata key is otherwise required. |
|
||||||
|
| `segments` | Yes | Non-empty array of segment objects, kept in input order. |
|
||||||
|
| `segments[].id` | Yes | Positive canonical decimal integer, supplied as a JSON number or string. IDs must be unique. |
|
||||||
|
| `segments[].start` | Yes | Finite, non-negative numeric value, supplied as a JSON number or string. |
|
||||||
|
| `segments[].end` | Yes | Finite, non-negative numeric value that is not earlier than `start`. |
|
||||||
|
| `segments[].speaker` | Yes | String that is non-empty after trimming. |
|
||||||
|
| `segments[].text` | Yes | String that is non-empty after trimming. Its original text is retained. |
|
||||||
|
|
||||||
Multiple top-level JSON values are rejected.
|
Additional top-level and segment fields are ignored. A missing required field,
|
||||||
|
`null` in place of an object or array, malformed JSON, or more than one
|
||||||
|
top-level JSON value is rejected.
|
||||||
|
|
||||||
## Validation
|
## Source Identity And References
|
||||||
|
|
||||||
The adapter rejects:
|
The adapter chooses the source ID in this order:
|
||||||
|
|
||||||
- empty raw input;
|
1. a non-empty source ID supplied by the calling request;
|
||||||
- malformed JSON;
|
2. non-empty string `metadata.id`;
|
||||||
- top-level JSON that is not an object;
|
3. non-empty string `metadata.source_id`;
|
||||||
- missing, null, or non-object `metadata`;
|
4. `seriatim:` followed by the first 16 hexadecimal characters of the raw
|
||||||
- missing, null, non-array, or empty `segments`;
|
input’s SHA-256 digest.
|
||||||
- segment values that are not objects;
|
|
||||||
- segment `id` values that are neither strings nor numbers;
|
|
||||||
- non-string `speaker` or `text`;
|
|
||||||
- empty segment IDs;
|
|
||||||
- segment IDs with leading or trailing whitespace;
|
|
||||||
- duplicate segment IDs;
|
|
||||||
- missing or empty `speaker`;
|
|
||||||
- missing, empty, invalid, non-finite, or negative `start`;
|
|
||||||
- missing, empty, invalid, non-finite, or negative `end`;
|
|
||||||
- `end` values before `start`;
|
|
||||||
- missing or empty `text`.
|
|
||||||
|
|
||||||
Segment text is preserved as provided, but it must not be empty after trimming.
|
Each accepted segment becomes one source unit whose unit ID is `segments[].id`.
|
||||||
|
Its self-reference uses the derived source ID and the same segment ID for both
|
||||||
|
range endpoints. Artifact contracts use those segment IDs when they cite
|
||||||
|
transcript evidence.
|
||||||
|
|
||||||
## Source Mapping
|
## Compatibility
|
||||||
|
|
||||||
The adapter maps input to `SourceDocument`:
|
This adapter accepts only the shape described here. A broader Seriatim export
|
||||||
|
is usable only when it supplies this object, metadata, and segment shape with
|
||||||
- `metadata` becomes `SourceDocument.Metadata`;
|
the stated types and constraints. Unknown additional fields do not add
|
||||||
- `SourceDocument.Kind` is `transcript`;
|
Notarius behavior.
|
||||||
- `SourceDocument.Format` is `application/vnd.seriatim+json`;
|
|
||||||
- `SourceDocument.Digest` is `sha256:<hex>` of the exact raw input bytes.
|
|
||||||
|
|
||||||
`SourceDocument.ID` is selected in this order:
|
|
||||||
|
|
||||||
1. the parse request source ID, after trimming;
|
|
||||||
2. `metadata.id`, when it is a non-empty string after trimming;
|
|
||||||
3. `metadata.source_id`, when it is a non-empty string after trimming;
|
|
||||||
4. `seriatim:<first-16-hex-chars-of-raw-sha256>`.
|
|
||||||
|
|
||||||
Each segment becomes one `SourceUnit`:
|
|
||||||
|
|
||||||
- `segment.id` becomes `SourceUnit.ID`; numeric IDs are converted to their JSON
|
|
||||||
number text, so `1` becomes `"1"`;
|
|
||||||
- `segment.text` becomes `SourceUnit.Text`;
|
|
||||||
- `SourceUnit.Kind` is `transcript_segment`;
|
|
||||||
- `speaker`, `start`, and `end` are stored in source-unit metadata.
|
|
||||||
|
|
||||||
## Metadata Keys
|
|
||||||
|
|
||||||
Seriatim unit metadata uses these keys:
|
|
||||||
|
|
||||||
- `speaker`: string speaker label;
|
|
||||||
- `start`: `json.Number` start value;
|
|
||||||
- `end`: `json.Number` end value.
|
|
||||||
|
|
||||||
The `internal/modules/input/seriatim` package exposes typed accessors for these
|
|
||||||
values.
|
|
||||||
|
|
||||||
## Capabilities
|
|
||||||
|
|
||||||
The module declares these provided capabilities:
|
|
||||||
|
|
||||||
- `source.transcript`
|
|
||||||
- `transcript.speaker`
|
|
||||||
- `transcript.timestamps`
|
|
||||||
|
|
||||||
## Compatibility Limit
|
|
||||||
|
|
||||||
This contract covers only Seriatim transcript JSON with the top-level
|
|
||||||
`metadata` object and `segments` array described here. Broader Seriatim output
|
|
||||||
schemas are compatible only when they provide these required fields with the
|
|
||||||
accepted types.
|
|
||||||
|
|||||||
145
docs/internal/cli.md
Normal file
145
docs/internal/cli.md
Normal file
@@ -0,0 +1,145 @@
|
|||||||
|
# CLI Internals
|
||||||
|
|
||||||
|
This document describes **internal/cli**, Notarius's production composition
|
||||||
|
root. The [CLI reference](../cli.md) owns command syntax and exit statuses;
|
||||||
|
[Configuration](../config.md) owns configuration values; and
|
||||||
|
[Operations](../operations.md) owns filesystem layout, recovery, and operator
|
||||||
|
procedures.
|
||||||
|
|
||||||
|
## Inputs, Outputs, And Boundaries
|
||||||
|
|
||||||
|
The CLI accepts process arguments, standard streams, and injectable options
|
||||||
|
used by tests and embedding code. It writes command results to the supplied
|
||||||
|
streams and returns a process exit status. For a run, it also creates the
|
||||||
|
production catalog and runtime collaborators, hands a prepared pipeline and
|
||||||
|
source bytes to the framework, and places the logical files returned by the
|
||||||
|
runner.
|
||||||
|
|
||||||
|
It is the only boundary allowed to compose concrete registries, LLM clients,
|
||||||
|
cache/checkpoint collaborators, debug recorders, and physical output paths.
|
||||||
|
Pipeline modules receive interfaces and request data rather than CLI streams or
|
||||||
|
filesystem roots. The [Architecture](../policy/architecture.md) defines this
|
||||||
|
composition-root boundary; [Pipeline Internals](pipeline.md) owns resolution,
|
||||||
|
preparation, and runner mechanics after their inputs are supplied.
|
||||||
|
|
||||||
|
## Dispatch And Configuration Handoff
|
||||||
|
|
||||||
|
The root dispatcher handles help, configuration validation, pipeline listing,
|
||||||
|
and a pipeline run. It normalizes injectable options before dispatch so that a
|
||||||
|
missing production dependency fails as a command error rather than reaching
|
||||||
|
execution.
|
||||||
|
|
||||||
|
Commands that need configuration use one shared loader. The CLI discovers the
|
||||||
|
file, parses it through **internal/core/config**, starts from defaults, applies
|
||||||
|
the file and supported environment overrides, and then validates it for the
|
||||||
|
command. The configured discovery and precedence contract is in
|
||||||
|
[Configuration](../config.md), while the loading and resolution mechanics are
|
||||||
|
in [Configuration Internals](configuration.md).
|
||||||
|
|
||||||
|
Configuration validation without a selected pipeline checks structural
|
||||||
|
configuration only. Validation with a selected pipeline also builds the
|
||||||
|
effective catalog, resolves the pipeline, and verifies explicitly selected
|
||||||
|
Scriptorium profiles. Pipeline listing validates configuration before returning
|
||||||
|
normalized, sorted identifiers.
|
||||||
|
|
||||||
|
## Production Composition
|
||||||
|
|
||||||
|
The production composition helper allocates every framework registry and the
|
||||||
|
prompt-asset registry, then registers the generic, Seriatim, and D&D module
|
||||||
|
families in that order. The resulting registries provide both the module
|
||||||
|
catalog used for resolution and the concrete constructors used for preparation.
|
||||||
|
Tests may provide a catalog or registries instead; production code must not
|
||||||
|
silently merge an injected partial catalog with production registrations.
|
||||||
|
|
||||||
|
The production LLM factory builds the Scriptorium-backed client from resolved
|
||||||
|
configuration, creates one scheduler from the effective global LLM limit, and
|
||||||
|
wraps the client before it reaches modules. Registration and LLM construction
|
||||||
|
errors are returned before a pipeline is prepared. Concrete module keys and
|
||||||
|
validator chains are public configuration choices and remain documented in
|
||||||
|
[Configuration](../config.md).
|
||||||
|
|
||||||
|
## Run Orchestration
|
||||||
|
|
||||||
|
After parsing and validating a run invocation, the CLI performs this ordered
|
||||||
|
handoff:
|
||||||
|
|
||||||
|
1. load and validate configuration, then apply command-level operational
|
||||||
|
overrides;
|
||||||
|
2. create and validate a safe run identity, then allocate a debug bundle only
|
||||||
|
when requested;
|
||||||
|
3. build the effective catalog, resolve requested reference changes, resolve
|
||||||
|
the effective pipeline, and verify explicit Scriptorium profiles;
|
||||||
|
4. materialize external or generated references and record redacted invocation
|
||||||
|
and resolution provenance when debug capture is enabled;
|
||||||
|
5. construct registries, the scheduled LLM client, prepared modules, and the
|
||||||
|
requested cache/checkpoint collaborators;
|
||||||
|
6. read the source input and invoke the framework runner; and
|
||||||
|
7. write the runner's logical output files only after a successful run, then
|
||||||
|
complete the command report and user-facing result.
|
||||||
|
|
||||||
|
Preparation happens before source parsing, so module construction and
|
||||||
|
dependency failures cannot begin stage execution. The CLI also preserves the
|
||||||
|
framework's result and warning information when it writes summaries and the
|
||||||
|
final command result. Detailed state lifecycle, resume handling, and physical
|
||||||
|
path confinement are maintained in [Run State Internals](state.md) and
|
||||||
|
[Operations](../operations.md).
|
||||||
|
|
||||||
|
For `run --json`, the CLI constructs and encodes its private run-result receipt
|
||||||
|
after a successful runner result is available, before it publishes logical
|
||||||
|
output files. It writes the prepared receipt to standard output only after
|
||||||
|
output publication and requested debug terminalization succeed. A receipt-write
|
||||||
|
failure exits with runtime status 1 and may leave partial standard-output bytes,
|
||||||
|
but the already-published output bundle remains complete and requested debug
|
||||||
|
reporting remains successfully terminalized. The CLI reports a bounded
|
||||||
|
command-owned error and does not repeat terminal reporting. The receipt remains
|
||||||
|
a CLI reporting concern rather than a framework or output-module responsibility;
|
||||||
|
its public contract is the
|
||||||
|
[run-result receipt](../integrations/run-result.md).
|
||||||
|
|
||||||
|
## Failure Mapping And Terminal Reporting
|
||||||
|
|
||||||
|
Argument, flag, and invocation-combination failures are reported to standard
|
||||||
|
error before runtime composition and use the syntax error class. Once an
|
||||||
|
invocation is syntactically valid, configuration loading and validation,
|
||||||
|
resolution, registration, profile checks, reference materialization, module
|
||||||
|
construction, input reads, runner failures, output publication, and requested
|
||||||
|
debug handling use the runtime failure class. The public status numbers and
|
||||||
|
stream contract are defined in the [CLI reference](../cli.md#output-streams-and-exit-statuses).
|
||||||
|
|
||||||
|
When debug capture has been allocated, one command-state value records the
|
||||||
|
known run result. Guarded terminalization writes a success report once, or
|
||||||
|
attempts a failure report and error record once. A persistence failure is
|
||||||
|
reported in addition to the original failure and never replaces it. If a debug
|
||||||
|
path exists, failure output includes that path so the retained diagnostic data
|
||||||
|
is discoverable.
|
||||||
|
|
||||||
|
## Invariants To Preserve
|
||||||
|
|
||||||
|
- Only the CLI composes production implementations and physical runtime roots.
|
||||||
|
- Configuration and resolved composition failures occur before module
|
||||||
|
preparation or source parsing.
|
||||||
|
- A runner's logical files are published only after a successful run.
|
||||||
|
- Production registries and a caller-supplied catalog or registries are
|
||||||
|
alternative composition sources, not an implicit mixture.
|
||||||
|
- A requested debug bundle has one terminal report attempt; its persistence
|
||||||
|
errors supplement rather than obscure the primary command error.
|
||||||
|
- User-facing flags, paths, exit codes, and configuration fields are defined
|
||||||
|
by their public documentation, not duplicated here.
|
||||||
|
|
||||||
|
## Focused Tests
|
||||||
|
|
||||||
|
- **internal/cli/command_contract_test.go** covers dispatch, help, syntax and
|
||||||
|
runtime error classes, discovery, validation, and listing.
|
||||||
|
- **internal/cli/run_contract_test.go** covers the run handoff, publication,
|
||||||
|
debug reporting, and command-owned state collaborators.
|
||||||
|
- **internal/cli/production_contract_test.go** covers registrar composition,
|
||||||
|
production catalog contents, assets, and representative configuration
|
||||||
|
validation.
|
||||||
|
- **internal/cli/reference_contract_test.go** covers CLI reference overrides,
|
||||||
|
origin separation, and materialization boundaries.
|
||||||
|
- **internal/cli/state_hardening_test.go** covers safe run identity, state
|
||||||
|
roots, and failure ordering.
|
||||||
|
|
||||||
|
Run **go test ./internal/cli** after changing command composition or command
|
||||||
|
behavior. Pair it with **go test ./internal/core/config** when the configuration
|
||||||
|
handoff changes.
|
||||||
129
docs/internal/configuration.md
Normal file
129
docs/internal/configuration.md
Normal file
@@ -0,0 +1,129 @@
|
|||||||
|
# Configuration Internals
|
||||||
|
|
||||||
|
This document describes the maintainer-facing configuration boundary in
|
||||||
|
**internal/core/config**. The [Configuration](../config.md) reference owns the
|
||||||
|
file format, fields, defaults, precedence contract, and selectable keys. The
|
||||||
|
[CLI reference](../cli.md) owns command syntax; this document does not redefine
|
||||||
|
either interface.
|
||||||
|
|
||||||
|
## Boundary
|
||||||
|
|
||||||
|
The configuration package turns a selected YAML file and supported environment
|
||||||
|
values into a validated, independently owned configuration. It then resolves a
|
||||||
|
requested pipeline against a module catalog before the framework prepares or
|
||||||
|
runs anything.
|
||||||
|
|
||||||
|
| Boundary | Inputs | Outputs | Does not own |
|
||||||
|
| --- | --- | --- | --- |
|
||||||
|
| Loading | Selected file path and environment lookup | Parsed file model and a populated **Config** | Choosing the file path or reporting a command result. |
|
||||||
|
| Validation | **Config** | Structural configuration errors with pipeline, lane, or binding context | Module availability, capabilities, or construction. |
|
||||||
|
| Resolution | Valid **Config**, selected pipeline and lanes, runtime reference changes, LLM override, and module catalog | **EffectiveConfig** with a **ResolvedPipeline** | Materializing reference bytes, preparing modules, execution, or filesystem state. |
|
||||||
|
| Summary | **Config** or **EffectiveConfig** | Detached redacted payload suitable for debug summaries | Redacting arbitrary process state or provider traffic. |
|
||||||
|
|
||||||
|
The CLI discovers a configuration file, invokes this package, and supplies the
|
||||||
|
result to the framework. Configuration never reads an input file, constructs a
|
||||||
|
module, or creates output, cache, or debug paths. Those responsibilities remain
|
||||||
|
at their respective [CLI](cli.md), [pipeline](pipeline.md), and
|
||||||
|
[run-state](state.md) boundaries.
|
||||||
|
|
||||||
|
## Loading And Validation
|
||||||
|
|
||||||
|
The CLI loads configuration in this order:
|
||||||
|
|
||||||
|
1. parse the selected YAML file strictly into the file model;
|
||||||
|
2. start from **Default**;
|
||||||
|
3. apply the file model; and
|
||||||
|
4. apply the supported environment overrides.
|
||||||
|
|
||||||
|
This establishes the public precedence order without giving environment input a
|
||||||
|
second file schema. Loading and application reject malformed YAML, unsupported
|
||||||
|
file versions, unknown fields, invalid values, and identifiers that are empty
|
||||||
|
or collide after whitespace normalization. The file application also makes the
|
||||||
|
effective extraction-worker default follow the effective LLM limit.
|
||||||
|
|
||||||
|
**Config.Validate** checks configuration-only invariants before resolution. It
|
||||||
|
rejects incompatible profile sources, invalid state-surface values, unsupported
|
||||||
|
concurrency settings, malformed bindings and references, invalid retries, and
|
||||||
|
invalid pipeline, step, or lane structure. Its errors retain the closest known
|
||||||
|
pipeline, lane, and binding context. It deliberately does not require modules
|
||||||
|
to be registered: that requires a catalog and belongs to resolution.
|
||||||
|
|
||||||
|
The exact user-selectable values and validation rules are defined in
|
||||||
|
[Configuration](../config.md). Keep additions to the file model, an
|
||||||
|
environment override, its validation, and that reference in the same change.
|
||||||
|
|
||||||
|
## Effective Resolution
|
||||||
|
|
||||||
|
**Config.Resolve** first recomputes derived concurrency defaults and validates
|
||||||
|
the configuration. It normalizes the requested pipeline ID, copies the selected
|
||||||
|
profile, applies a non-empty command-level LLM profile override to the
|
||||||
|
LLM-capable stage bindings, and calls the framework resolver with the requested
|
||||||
|
lane selection and reference changes.
|
||||||
|
|
||||||
|
The command-level override does not replace an explicitly selected validator
|
||||||
|
profile. Validator bindings remain part of the resolved validator chain and
|
||||||
|
are resolved under their own declared configuration.
|
||||||
|
|
||||||
|
The framework resolver supplies defaults, selects lanes, resolves validator
|
||||||
|
chains, checks registered module and artifact compatibility, validates module
|
||||||
|
options, and returns the fixed ordered pipeline shape. The resulting
|
||||||
|
**EffectiveConfig** retains the selected ID, requested selection and reference
|
||||||
|
changes, a clone of the input configuration, and the resolved pipeline.
|
||||||
|
Callers may therefore retain or modify their input slices and maps without
|
||||||
|
changing the resolved result, and later consumers cannot mutate the original
|
||||||
|
configuration through the effective value.
|
||||||
|
|
||||||
|
Resolution failures stop before module construction and source parsing. They
|
||||||
|
include an error path for an unconfigured pipeline, missing module, missing
|
||||||
|
capability, incompatible artifact variant, invalid option, invalid reference,
|
||||||
|
or invalid lane selection. CLI code maps these valid-invocation failures to the
|
||||||
|
runtime error class described in the [CLI reference](../cli.md#output-streams-and-exit-statuses).
|
||||||
|
|
||||||
|
## Resolved Identity And Redaction
|
||||||
|
|
||||||
|
The framework assigns the resolved pipeline a deterministic SHA-256 digest
|
||||||
|
after defaults, lane selection, module bindings, reference bindings, validator
|
||||||
|
chains, and artifact schema identity have been resolved. The digest excludes
|
||||||
|
its own stored value. It identifies resolved composition rather than raw YAML
|
||||||
|
bytes, a debug payload, or all runtime state. The CLI records it as invocation
|
||||||
|
provenance before execution; cache and checkpoint identity have additional
|
||||||
|
owners in [Run State Internals](state.md).
|
||||||
|
|
||||||
|
Configuration summaries must use **Redacted**, **RedactedSummaryPayload**, or
|
||||||
|
**RedactedResolvedPipelinePayload**, never a direct configuration marshal.
|
||||||
|
Those methods copy every binding and nested option container, replace values
|
||||||
|
whose key is credential-shaped with **[REDACTED]**, and omit materialized
|
||||||
|
reference content while retaining safe binding and reference provenance. The
|
||||||
|
payload must not alias the source configuration or resolved pipeline. This
|
||||||
|
redaction is deliberately narrow: it protects configuration summaries and does
|
||||||
|
not authorize recording arbitrary environment values or provider requests.
|
||||||
|
|
||||||
|
## Invariants To Preserve
|
||||||
|
|
||||||
|
- Defaults, YAML values, and environment values are applied in one direction;
|
||||||
|
later sources may override only their supported operational settings.
|
||||||
|
- A configuration is structurally valid before it is resolved, and a resolved
|
||||||
|
pipeline is compatible with the supplied catalog before preparation begins.
|
||||||
|
- Whitespace-normalized identifiers are unique wherever they identify a
|
||||||
|
pipeline, step, lane, worker, or reference slot.
|
||||||
|
- Resolution and summary generation return detached data. Redaction must cover
|
||||||
|
every configured and resolved binding, including nested validator bindings.
|
||||||
|
- The resolved digest changes when resolved composition changes and never
|
||||||
|
includes itself.
|
||||||
|
|
||||||
|
## Focused Tests
|
||||||
|
|
||||||
|
- **internal/core/config/file_config_contract_test.go** covers strict file
|
||||||
|
parsing, normalization, file application, and structural rejection.
|
||||||
|
- **internal/core/config/env_contract_test.go** covers supported operational
|
||||||
|
overrides and their precedence.
|
||||||
|
- **internal/core/config/validation_contract_test.go** covers configuration
|
||||||
|
invariants and contextual failures.
|
||||||
|
- **internal/core/config/effective_config_contract_test.go** covers defaults,
|
||||||
|
selections, overrides, resolution context, digest changes, and ownership.
|
||||||
|
- **internal/core/config/redaction_test.go** covers recursive credential
|
||||||
|
redaction, reference-content exclusion, and non-aliasing payloads.
|
||||||
|
|
||||||
|
Run **go test ./internal/core/config** after changing this boundary. Changes to
|
||||||
|
the handoff or resolved-composition semantics also need the focused framework
|
||||||
|
pipeline tests.
|
||||||
@@ -1,88 +0,0 @@
|
|||||||
# Diagnostics Internals
|
|
||||||
|
|
||||||
Diagnostics internals live in `internal/core/diagnostics`. Operator-facing run
|
|
||||||
behavior is documented in [Operations](../operations.md).
|
|
||||||
|
|
||||||
## Purpose
|
|
||||||
|
|
||||||
Diagnostics provide local inspection artifacts for a run without becoming the
|
|
||||||
durable output contract. Durable user output is produced by output modules and
|
|
||||||
written by the CLI.
|
|
||||||
|
|
||||||
Diagnostics must not expose secrets.
|
|
||||||
|
|
||||||
## Run Directory
|
|
||||||
|
|
||||||
`NewRunDirectory(workDir, retention)` creates:
|
|
||||||
|
|
||||||
```text
|
|
||||||
<workDir>/run-<unix-nanoseconds>/
|
|
||||||
```
|
|
||||||
|
|
||||||
If `workDir` is empty, it defaults to `/tmp/notarius`. Empty retention defaults
|
|
||||||
to `auto`.
|
|
||||||
|
|
||||||
The writer makes the work directory if needed, then attempts to create a unique
|
|
||||||
run directory. It retries run ID creation a bounded number of times if a
|
|
||||||
collision occurs.
|
|
||||||
|
|
||||||
## Artifact Writers
|
|
||||||
|
|
||||||
Implemented artifact names:
|
|
||||||
|
|
||||||
- `invocation.json`
|
|
||||||
- `effective-config.json`
|
|
||||||
- `resolved-pipeline.json`
|
|
||||||
- `source-document.json`
|
|
||||||
- `run-manifest.json`
|
|
||||||
- `run-report.json`
|
|
||||||
- `warnings.json`
|
|
||||||
- `error.log`
|
|
||||||
|
|
||||||
JSON artifacts are encoded with indentation and a trailing newline. Writes are
|
|
||||||
atomic through a temporary file in the target directory followed by rename.
|
|
||||||
|
|
||||||
Artifact names must be single relative file names. Absolute paths, path
|
|
||||||
separators, and names resolving outside the run directory are rejected.
|
|
||||||
|
|
||||||
## Redacted Effective Config
|
|
||||||
|
|
||||||
Diagnostics writers accept payloads that implement
|
|
||||||
`RedactedDiagnosticsPayload`. `internal/core/config` uses this to redact API
|
|
||||||
keys in effective config diagnostics while preserving resolved pipeline context.
|
|
||||||
|
|
||||||
The redaction path clones config data before replacing secret values.
|
|
||||||
|
|
||||||
## Retention
|
|
||||||
|
|
||||||
Retention is decided by `ShouldRetainRunDirectory`.
|
|
||||||
|
|
||||||
- Failed runs are always retained.
|
|
||||||
- `always` retains successful runs.
|
|
||||||
- `never` removes successful runs.
|
|
||||||
- `auto` retains successful runs only when warnings exist.
|
|
||||||
- Unknown retention values are treated as retain by the retention decision, but
|
|
||||||
config validation rejects unsupported values before normal runs.
|
|
||||||
|
|
||||||
`ApplyRetention` removes only the specific run directory.
|
|
||||||
|
|
||||||
## CLI Failure Behavior
|
|
||||||
|
|
||||||
The CLI creates the diagnostics run directory after config loading and before
|
|
||||||
pipeline resolution. Failures before that point do not have diagnostics.
|
|
||||||
|
|
||||||
After diagnostics creation, run failures call `WriteErrorLog` and apply
|
|
||||||
retention with `RunSucceeded: false`, so the run directory remains available.
|
|
||||||
|
|
||||||
When the pipeline returns a partial manifest on failure, the CLI writes that
|
|
||||||
manifest before logging the failure.
|
|
||||||
|
|
||||||
## Invariants
|
|
||||||
|
|
||||||
- Diagnostics paths must be narrow and run-directory scoped.
|
|
||||||
- Writes should be atomic where practical.
|
|
||||||
- Secrets must be redacted.
|
|
||||||
- Diagnostics write failures are command failures because they can hide the
|
|
||||||
information needed for recovery.
|
|
||||||
- Durable output file contracts belong to output modules and integration docs,
|
|
||||||
not to diagnostics.
|
|
||||||
119
docs/internal/dnd.md
Normal file
119
docs/internal/dnd.md
Normal file
@@ -0,0 +1,119 @@
|
|||||||
|
# D&D Module Internals
|
||||||
|
|
||||||
|
This guide records the conventions shared by the production D&D module family.
|
||||||
|
It complements [Module Internals](modules.md), which owns generic registration
|
||||||
|
and extension mechanics, and [Configuration](../config.md), which owns the
|
||||||
|
selectable keys, bindings, reference syntax, and default validator chains.
|
||||||
|
|
||||||
|
## Durable Artifact Contracts
|
||||||
|
|
||||||
|
The six lanes have separate durable wire contracts. This guide deliberately
|
||||||
|
does not repeat their JSON shapes or schemas.
|
||||||
|
|
||||||
|
| Lane | Durable contract |
|
||||||
|
| --- | --- |
|
||||||
|
| Spells | [spell artifacts](../integrations/dnd-spell-artifacts.md) |
|
||||||
|
| NPCs | [NPC artifacts](../integrations/dnd-npc-artifacts.md) |
|
||||||
|
| Combat turns | [combat-turn artifacts](../integrations/dnd-combat-turn-artifacts.md) |
|
||||||
|
| Item events | [item-event artifacts](../integrations/dnd-item-event-artifacts.md) |
|
||||||
|
| NPC interactions | [NPC-interaction artifacts](../integrations/dnd-npc-interaction-artifacts.md) |
|
||||||
|
| Scene descriptions | [scene-description artifacts](../integrations/dnd-scene-description-artifacts.md) |
|
||||||
|
|
||||||
|
## Family Composition
|
||||||
|
|
||||||
|
The D&D registrar registers the family’s artifact codecs, extractors, typed
|
||||||
|
append-order mergers, normalizers, validators, prompt assets, and default
|
||||||
|
validator chains. Each extractor and normalizer has a stable module spec,
|
||||||
|
strict option decoding, and a typed builder. Configuration remains the
|
||||||
|
canonical owner of the exact keys and validator order.
|
||||||
|
|
||||||
|
Private structured-LLM response schemas are deliberately minimal. They reject
|
||||||
|
invalid JSON structure, missing required fields, incompatible types, and
|
||||||
|
unknown fields, while preserving semantic candidates for deterministic
|
||||||
|
validation. Do not promote a private response envelope into a durable schema;
|
||||||
|
the contracts above define durable data.
|
||||||
|
|
||||||
|
## Prompt Construction
|
||||||
|
|
||||||
|
D&D extractors assemble prompts from an ordered manifest of shared and
|
||||||
|
module-owned assets. Reuse the shared D&D system, evidence, identity,
|
||||||
|
reference, and transcript assets instead of copying their text into individual
|
||||||
|
modules. A manifest’s declared sequence, including cache-control placement, is
|
||||||
|
part of the prompt behavior, and the chunk transcript is the final message.
|
||||||
|
Preserve that order when changing an extractor or its assets so prompt-cache
|
||||||
|
behavior remains stable.
|
||||||
|
|
||||||
|
All extractors use the shared prompt-input preparation rules. The current chunk
|
||||||
|
is copied into transcript material; player, party, glossary, and compatible
|
||||||
|
campaign references are context for disambiguation, not source evidence.
|
||||||
|
Reference prompt material is canonically ordered before it is rendered, which
|
||||||
|
keeps equivalent inputs stable across runs.
|
||||||
|
|
||||||
|
## Evidence, Candidates, And Normalization
|
||||||
|
|
||||||
|
The current transcript is the only durable evidence source. Extractors assign
|
||||||
|
the current source identity, preserve candidate evidence ranges for validators,
|
||||||
|
and canonically order or remove exact duplicate ranges without asking the
|
||||||
|
model to repair semantic errors. Campaign context and generated artifacts may
|
||||||
|
ground names or control routing, but they never establish evidence for a D&D
|
||||||
|
result.
|
||||||
|
|
||||||
|
Default chains keep responsibilities separate: structural validators assess the
|
||||||
|
candidate, source-reference validators resolve cited ranges against the current
|
||||||
|
source, durable-schema validation checks an approved representation, and
|
||||||
|
relatedness validators report advisory evidence concerns. The configured order
|
||||||
|
is documented in
|
||||||
|
[Configuration](../config.md#production-validator-keys-and-default-chains).
|
||||||
|
|
||||||
|
Normalizers are deterministic for spells, combat turns, item events, NPC
|
||||||
|
interactions, and scene descriptions. They canonicalize display values and
|
||||||
|
evidence, use source-document order for stable output, and issue bounded
|
||||||
|
warnings for changes or collapsed duplicates. The NPC normalizer is the
|
||||||
|
intentional exception: it first produces a deterministic candidate set, then
|
||||||
|
uses a bounded structured-LLM proposal to reconcile identity groups. Invalid
|
||||||
|
or unusable proposals retain the deterministic result and surface retry or
|
||||||
|
fallback diagnostics; the model does not directly replace durable records.
|
||||||
|
|
||||||
|
## Generated References And Grounding
|
||||||
|
|
||||||
|
Normalized D&D artifacts can be handed to a later step through a generated
|
||||||
|
reference binding. The framework verifies artifact compatibility and retains
|
||||||
|
producer provenance; consumers resolve the handed-off artifact into an
|
||||||
|
immutable, validated projection for each operation. External files are checked
|
||||||
|
during preparation, while generated artifacts are resolved at the handoff.
|
||||||
|
|
||||||
|
NPC registries are names-only grounding projections: they may canonicalize
|
||||||
|
actors for spells and combat turns and are required for NPC interactions, but
|
||||||
|
they do not supply evidence. Scene-description registries are eligibility-only
|
||||||
|
projections: they retain the current chunk’s classification data, not scene
|
||||||
|
prose or evidence, and exist to route combat extraction.
|
||||||
|
|
||||||
|
## Lane-Specific Rules
|
||||||
|
|
||||||
|
The following differences are intentional and should remain explicit when a
|
||||||
|
shared helper changes.
|
||||||
|
|
||||||
|
| Lane | Intentional behavior |
|
||||||
|
| --- | --- |
|
||||||
|
| Spells | May use a spell-catalog overlay and optional NPC grounding; the catalog validator supplies domain-specific semantic checks. |
|
||||||
|
| NPCs | Does not consume an NPC registry. Its normalizer is the LLM-assisted reconciliation exception described above. |
|
||||||
|
| Combat turns | Requires a scene-description artifact. It calls the LLM only for an exact `combat` classification; exact non-combat classifications return an accepted empty result, while missing or mismatched classifications return an empty result with a bounded warning. Optional NPC grounding never becomes evidence. |
|
||||||
|
| Item events | Uses campaign context for disambiguation but has no NPC-registry or scene-description dependency. |
|
||||||
|
| NPC interactions | Requires the normalized NPC registry at extraction and normalization, using it for canonical actor grounding only. |
|
||||||
|
| Scene descriptions | Produces the classifications consumed by combat routing; it does not consume an NPC registry or provide evidence for combat artifacts. |
|
||||||
|
|
||||||
|
The combat and scene-description contracts describe their exact handoff and
|
||||||
|
empty-result behavior in more detail:
|
||||||
|
[combat turns](../integrations/dnd-combat-turn-artifacts.md) and
|
||||||
|
[scene descriptions](../integrations/dnd-scene-description-artifacts.md).
|
||||||
|
|
||||||
|
## Focused Verification
|
||||||
|
|
||||||
|
When changing D&D behavior, test the affected codec, extractor, normalizer,
|
||||||
|
validator, prompt-asset manifest, and registry projection. Also test generated
|
||||||
|
handoffs at the integration boundary and run the full D&D module suite:
|
||||||
|
|
||||||
|
~~~sh
|
||||||
|
go test ./internal/modules/dnd/...
|
||||||
|
go test ./internal/modules/integration/...
|
||||||
|
~~~
|
||||||
@@ -1,116 +1,149 @@
|
|||||||
# LLM Runtime
|
# LLM Runtime Internals
|
||||||
|
|
||||||
The implemented LLM runtime lives in `internal/framework/llm`. It provides
|
`internal/framework/llm` is Notarius’s provider-independent structured
|
||||||
transport-neutral structured completion contracts, an OpenAI-compatible HTTP
|
completion boundary. It adapts framework requests to Scriptorium, bounds
|
||||||
adapter, concurrency scheduling, schema registry helpers, retry behavior, and
|
provider calls, assembles registered prompt and schema assets, records selected
|
||||||
secret redaction.
|
profiles, and redacts provider errors. The architectural boundary is defined in
|
||||||
|
[Architecture](../policy/architecture.md#llm-boundary); profile sources,
|
||||||
|
credentials, and concurrency settings belong in
|
||||||
|
[Configuration](../config.md#scriptorium-profiles) and
|
||||||
|
[Configuration](../config.md#concurrency-output-cache-and-debug).
|
||||||
|
|
||||||
## Contract
|
## Structured Completion Boundary
|
||||||
|
|
||||||
Modules depend on `contracts.StructuredLLMClient`:
|
Modules and LLM-backed validators depend only on
|
||||||
|
`contracts.StructuredLLMClient`. A completion request supplies a prompt ID and
|
||||||
|
version, optional profile and session IDs, named input material, variables, and
|
||||||
|
a caller-owned decode target. The successful response returns the validated raw
|
||||||
|
structured bytes together with non-secret provider, model, profile, and token
|
||||||
|
metadata.
|
||||||
|
|
||||||
```go
|
The caller owns the domain behavior: it chooses the prompt, prepares inputs,
|
||||||
CompleteStructured(ctx, request, out) (response, error)
|
selects the private response schema, and interprets the decoded result. The
|
||||||
```
|
adapter does not own source evidence, artifact conversion, normalization, or
|
||||||
|
durable schemas. Those responsibilities remain with the module and its
|
||||||
|
[integration contract](../integrations/).
|
||||||
|
|
||||||
The request contains messages, optional model override, response schema name,
|
`ScriptoriumClient` validates the request target and prompt identity, maps each
|
||||||
and response schema JSON. The caller supplies a pointer target for decoded
|
named material to a Scriptorium inline artifact while preserving its origin URI,
|
||||||
structured output.
|
forwards session and profile selection, then prepares and runs the prompt. It
|
||||||
|
returns Scriptorium’s validated raw bytes rather than re-encoding the decoded
|
||||||
|
target. An empty optional material is represented as one space so its named
|
||||||
|
input is retained by Scriptorium.
|
||||||
|
|
||||||
Extractors own prompts and schemas. Provider adapters should not contain
|
An empty request profile lets the prompt select its configured default. The CLI
|
||||||
domain-specific prompt logic.
|
prepares every explicitly selected binding profile before a run begins, so a
|
||||||
|
missing explicit profile fails before stage execution. Calls record the profile
|
||||||
|
actually selected by Scriptorium; the recorder deduplicates non-secret profile
|
||||||
|
identity, provider, and model values for manifest use.
|
||||||
|
|
||||||
## Production Client Construction
|
## Shared Provider-Call Limit
|
||||||
|
|
||||||
`internal/cli` builds the production LLM client from the effective config:
|
Production construction creates one Scriptorium client and wraps it in one
|
||||||
|
scheduled client. The scheduler has a fixed, positive permit limit, serves
|
||||||
|
queued calls in FIFO order, and removes a queued call when its context is
|
||||||
|
cancelled. A granted permit is released exactly once on every completion path.
|
||||||
|
|
||||||
1. find the effective LLM profile;
|
The scheduled wrapper surrounds every `CompleteStructured` call, so concurrent
|
||||||
2. build `OpenAICompatibleClientConfig`;
|
lanes, pipeline retries, and LLM-backed validators share the same provider-call
|
||||||
3. create an OpenAI-compatible client;
|
ceiling. This ceiling is independent of pipeline worker concurrency; changing
|
||||||
4. create a scheduler from profile or global concurrency;
|
worker counts cannot exceed the configured LLM limit. The configuration field
|
||||||
5. wrap the client with `NewScheduledClient`;
|
and its effective default are owned by
|
||||||
6. return non-secret LLM profile manifest metadata.
|
[Configuration](../config.md#concurrency-output-cache-and-debug).
|
||||||
|
|
||||||
The current run command requires exactly one distinct effective LLM profile for
|
## Prompt And Schema Assets
|
||||||
the resolved pipeline.
|
|
||||||
|
|
||||||
## OpenAI-Compatible Adapter
|
An `AssetRegistry` collects prompt and schema filesystems from production module
|
||||||
|
families. It flattens registered roots into the Scriptorium filesystems and
|
||||||
|
rejects invalid roots, unreadable assets, duplicate paths, and missing prompt
|
||||||
|
or schema files during preparation. The framework’s `promptfs` helper combines
|
||||||
|
module-owned prompt files with reusable domain fragments without making the
|
||||||
|
framework depend on D&D content.
|
||||||
|
|
||||||
`OpenAICompatibleClient` posts JSON to:
|
Each LLM-backed module owns its prompt declaration, package-specific assets,
|
||||||
|
and private response schema. Shared D&D wording is owned by the D&D shared
|
||||||
|
asset package; the detailed D&D conventions are in
|
||||||
|
[D&D Module Internals](dnd.md). The mounted prompt assets used by a module also
|
||||||
|
determine its prompt fingerprint. Schema loaders validate JSON, attach identity
|
||||||
|
and digest metadata, make defensive copies, and expose diagnostics without raw
|
||||||
|
schema bytes.
|
||||||
|
|
||||||
```text
|
Private response schemas validate a model transport envelope. They are not the
|
||||||
<base_url>/chat/completions
|
durable artifact schema and should not be documented as an external wire
|
||||||
```
|
contract. Durable formats and compatibility rules remain in the
|
||||||
|
[integration contracts](../integrations/).
|
||||||
|
|
||||||
It sends:
|
## Prompt Maintenance And Backend Caching
|
||||||
|
|
||||||
- `model`
|
Prompt message order and shared asset bytes are runtime behavior. Backend cache
|
||||||
- `messages`
|
reuse depends on the same preceding messages and content, not merely equivalent
|
||||||
- `response_format.type = "json_schema"`
|
meaning. Keep reusable shared assets byte-identical and keep stable material
|
||||||
- `response_format.json_schema.name`
|
before the inputs that vary per request wherever a prompt’s declared sequence
|
||||||
- `response_format.json_schema.strict = true`
|
supports caching. Preserve the existing manifest order and cache-control hints
|
||||||
- `response_format.json_schema.schema`
|
when editing a prompt.
|
||||||
|
|
||||||
If an API key is configured, the adapter sends an `Authorization: Bearer ...`
|
D&D extraction manifests place the changing chunk transcript at the end of the
|
||||||
header.
|
prompt after their reusable context. Scene chunking and NPC normalization use
|
||||||
|
their own declared message sequences because their inputs and work differ. The
|
||||||
|
family-specific asset and ordering rules belong in [D&D Module Internals](dnd.md).
|
||||||
|
Do not add tests that enforce a fixed message-prefix length; prompt-asset tests
|
||||||
|
should instead verify the meaningful asset sequence, inputs, and cache controls
|
||||||
|
of the prompt being changed.
|
||||||
|
|
||||||
The adapter accepts assistant content either as a JSON string containing JSON or
|
## Validation, Repair, And Retries
|
||||||
as raw JSON content. It then unmarshals that content into the caller-provided
|
|
||||||
target.
|
|
||||||
|
|
||||||
External wire-contract details belong in the
|
Scriptorium performs prompt rendering, provider execution, and the prompt’s
|
||||||
[OpenAI-compatible integration doc](../integrations/openai-compatible.md).
|
structured-output validation. The adapter reports an empty result, validation
|
||||||
|
failure, empty structured body, or decode failure as
|
||||||
|
`ErrInvalidStructuredOutput`, while retaining the returned raw bytes and debug
|
||||||
|
material when they exist. Provider failures remain operational errors rather
|
||||||
|
than output-validation failures.
|
||||||
|
|
||||||
## Retries And Timeouts
|
Prompt-declared repair is executed within Scriptorium’s structured-output flow.
|
||||||
|
The current production D&D prompt manifests set repair attempts to zero. That
|
||||||
|
setting does not replace pipeline retry behavior: a binding’s configured retry
|
||||||
|
count reruns its stage attempt after an error or rejection, and an exhausted
|
||||||
|
rejection is a recorded output rather than a provider error. The pipeline owns
|
||||||
|
attempt lifecycle, validation chains, and retry diagnostics; see
|
||||||
|
[Pipeline Internals](pipeline.md#validation-retries-and-output) and the
|
||||||
|
[binding reference](../config.md#module-bindings-and-validators).
|
||||||
|
|
||||||
The adapter retries:
|
## Observability And Redaction
|
||||||
|
|
||||||
- provider request failures;
|
When debug recording is enabled, the pipeline decorates the shared client. The
|
||||||
- response read failures;
|
wrapper records prepared prompt and response material, timing, selected profile
|
||||||
- HTTP `429`;
|
and model, and call identifiers in the run’s debug bundle, including material
|
||||||
- HTTP `5xx`;
|
available from a failed structured completion. For a successful completion, a
|
||||||
- malformed provider envelopes;
|
debug-write failure is surfaced; when the completion already failed, its call
|
||||||
- malformed assistant JSON;
|
error remains the result. Debug-bundle location, retention, and handling are
|
||||||
- structured-output decode failures.
|
operational concerns documented in [Operations](../operations.md#debug-bundles).
|
||||||
|
|
||||||
Non-retryable `4xx` responses are returned without retry. Request timeout comes
|
Run manifests receive selected profile summaries and component identities, not
|
||||||
from the effective LLM profile. Context cancellation is respected.
|
prompt, schema, source, reference, or response content. Provider error text is
|
||||||
|
wrapped with prompt context and bearer credentials are redacted before it
|
||||||
|
crosses the runtime boundary. Known-secret redaction is available to other
|
||||||
|
runtime collaborators; it does not make prompt or response contents safe for
|
||||||
|
general logging.
|
||||||
|
|
||||||
## Scheduler
|
## Failure Boundaries
|
||||||
|
|
||||||
`Scheduler` bounds concurrent provider calls. It tracks in-flight calls and a
|
- Construction fails for missing asset registries, mutually exclusive profile
|
||||||
FIFO queue of waiters. Cancellation removes queued waiters or releases granted
|
sources, invalid asset registration, or a non-positive scheduler limit.
|
||||||
permits.
|
- Preparation failures, unavailable explicit profiles, provider failures, and
|
||||||
|
context cancellation propagate to the calling stage with context.
|
||||||
|
- Malformed or schema-invalid provider output is classified separately as
|
||||||
|
invalid structured output so the module or pipeline can apply its own retry
|
||||||
|
and rejection policy.
|
||||||
|
- Domain semantic checks, evidence decisions, and deterministic normalization
|
||||||
|
run outside the provider adapter.
|
||||||
|
|
||||||
`NewScheduledClient` wraps any structured LLM client and runs each completion
|
## Focused Verification
|
||||||
inside the scheduler.
|
|
||||||
|
|
||||||
Effective concurrency is:
|
Read the LLM adapter, scheduler, asset registry, schema loader, and redaction
|
||||||
|
tests when changing this boundary. Prompt changes also require the owning
|
||||||
|
module’s asset tests, and retry or debug changes require focused pipeline or
|
||||||
|
CLI coverage. The focused runtime and D&D checks are:
|
||||||
|
|
||||||
1. `llm_profiles.<id>.max_concurrency`, when greater than zero;
|
~~~sh
|
||||||
2. `concurrency.total_llm`, when greater than zero;
|
go test ./internal/framework/llm/... ./internal/modules/dnd/...
|
||||||
3. `1`.
|
~~~
|
||||||
|
|
||||||
## Schema Registry
|
|
||||||
|
|
||||||
The framework schema registry embeds generic test schemas. It also exposes
|
|
||||||
helpers for caller-owned schemas:
|
|
||||||
|
|
||||||
- `LoadResponseSchema`
|
|
||||||
- `LookupResponseSchema`
|
|
||||||
- `MustLookupResponseSchema`
|
|
||||||
- `ResponseSchema.DiagnosticsMap`
|
|
||||||
|
|
||||||
`DiagnosticsMap` omits raw schema content and includes metadata such as key,
|
|
||||||
ID, version, name, and SHA-256.
|
|
||||||
|
|
||||||
The D&D spell extractor owns and loads its own embedded response schema.
|
|
||||||
|
|
||||||
## Secret Redaction
|
|
||||||
|
|
||||||
Provider errors are passed through `ErrorWithSecretsRedacted` with the API key
|
|
||||||
and bearer-token value. Config diagnostics use redacted effective config
|
|
||||||
payloads.
|
|
||||||
|
|
||||||
Do not add raw provider request bodies, response bodies, API keys, or prompt
|
|
||||||
payloads to diagnostics by default.
|
|
||||||
|
|||||||
@@ -1,165 +1,107 @@
|
|||||||
# Modules
|
# Module Internals
|
||||||
|
|
||||||
Production modules live under `internal/modules`. Each module implements one
|
This guide owns the mechanics for implementing and registering production
|
||||||
contract from `internal/framework/contracts`, exposes a `ModuleSpec`, and
|
modules. [Configuration](../config.md) owns selectable keys, binding syntax,
|
||||||
registers itself with the matching pipeline registry.
|
reference configuration, and default validator chains. Durable input and output
|
||||||
|
shapes belong in [integration contracts](../integrations/).
|
||||||
|
|
||||||
The CLI production catalog currently registers only the modules listed here.
|
The D&D family has additional shared conventions and domain-specific
|
||||||
|
exceptions. See [D&D Module Internals](dnd.md) rather than adding them here.
|
||||||
|
|
||||||
## Contract Pattern
|
## Module Boundary
|
||||||
|
|
||||||
A production module package should provide:
|
A module is a typed implementation registered for one pipeline stage. Its
|
||||||
|
`ModuleSpec` is the public-to-the-framework declaration of its stable key,
|
||||||
|
stage, required and provided capabilities, artifact kind, and accepted
|
||||||
|
reference slots. The framework uses that declaration to resolve a configured
|
||||||
|
binding before it builds the implementation.
|
||||||
|
|
||||||
- a stable module key;
|
Implementations that accept options must provide both an option validator and
|
||||||
- a constructor such as `New`;
|
a builder. The validator is used while resolving configuration; the builder
|
||||||
- the relevant contract implementation;
|
decodes the same options and constructs the implementation from the prepared
|
||||||
- `ModuleSpec`;
|
`BuildRequest`. Reject unknown options in both paths. A builder receives only
|
||||||
- `Register`;
|
the dependencies and materialized references that the framework prepared for
|
||||||
- focused tests for registration, options, contract behavior, and errors.
|
that operation, so it must not re-read configuration or files.
|
||||||
|
|
||||||
Module specs should describe capabilities accurately. Resolution uses specs to
|
Registry helpers register the typed builder for a stage-specific registry.
|
||||||
reject incompatible pipelines before execution.
|
They are preferable to hand-written untyped registration because they retain
|
||||||
|
the artifact type at the framework boundary. Registrars validate the registries
|
||||||
|
they need, register each leaf implementation, and add any family-owned assets
|
||||||
|
or default validator chains. They return contextual errors so production
|
||||||
|
composition fails at startup rather than at the first run.
|
||||||
|
|
||||||
## `seriatim` Input
|
An artifact family can register an optional typed evidence projector alongside
|
||||||
|
its codec. The projector returns defensive copies of the artifact's direct
|
||||||
|
generic source references and must use the codec's exact Go type. It does not
|
||||||
|
interpret surrounding context or publish files; the pipeline validates the
|
||||||
|
capability during preparation and the output boundary owns publication. See
|
||||||
|
the [Published Evidence Context contract](../integrations/evidence-context.md)
|
||||||
|
for the durable result.
|
||||||
|
|
||||||
Package: `internal/modules/input/seriatim`
|
## Production Composition
|
||||||
|
|
||||||
The `seriatim` adapter parses Seriatim transcript JSON into a generic source
|
Production composition is intentionally split by family:
|
||||||
document. It owns transcript JSON details, source ID selection, source digest
|
|
||||||
creation, transcript segment validation, and segment metadata mapping.
|
|
||||||
|
|
||||||
Provides:
|
- The generic registrar provides the unit chunker, generic JSON validators,
|
||||||
|
and JSON output encoder.
|
||||||
|
- The Seriatim registrar provides the transcript input adapter. Its external
|
||||||
|
input behavior is defined by the [Seriatim contract](../integrations/seriatim.md).
|
||||||
|
- The D&D registrar provides its codecs, extractors, mergers, normalizers,
|
||||||
|
validators, prompt assets, and default chains. Its behavioral conventions
|
||||||
|
are documented in [D&D Module Internals](dnd.md).
|
||||||
|
|
||||||
- `source.transcript`
|
The CLI owns the composition that invokes these registrars. A module package
|
||||||
- `transcript.speaker`
|
may register its own family but must not assemble the CLI or make framework
|
||||||
- `transcript.timestamps`
|
packages depend on production extensions.
|
||||||
|
|
||||||
External JSON shape belongs in the Seriatim integration doc.
|
## Adding Or Changing A Module
|
||||||
|
|
||||||
## `generic` Chunker
|
1. Choose the pipeline stage and the typed artifact boundary. Put external
|
||||||
|
input or durable artifact formats in the relevant integration contract,
|
||||||
|
not in this guide or in a private LLM response type.
|
||||||
|
2. Define a stable `ModuleSpec` with the exact capabilities and reference
|
||||||
|
slots needed for the operation. Model a producer/consumer handoff as an
|
||||||
|
artifact-compatible slot; configuration then chooses an external file or a
|
||||||
|
generated binding.
|
||||||
|
3. Implement strict option decoding, construction, and the typed stage
|
||||||
|
interface. Preserve caller ownership: do not retain mutable request data
|
||||||
|
and return defensive copies where an implementation exposes stored data.
|
||||||
|
4. Register the module through its typed registry helper and add it to the
|
||||||
|
owning family registrar. Add a default validator chain only when that
|
||||||
|
family owns the behavior; otherwise require an explicit compatible chain.
|
||||||
|
5. Update the selectable-key and chain reference in
|
||||||
|
[Configuration](../config.md#production-module-keys), the applicable
|
||||||
|
integration contract, and focused tests. Keep the configuration document
|
||||||
|
as the sole list of production keys and validator order.
|
||||||
|
|
||||||
Package: `internal/modules/chunk/generic`
|
## Validation And References
|
||||||
|
|
||||||
The `generic` chunker splits source units into ordered chunks. It validates the
|
Validators operate on the value produced at their configured stage. A default
|
||||||
source document, clones source units, assigns chunk IDs such as `chunk-000001`,
|
chain is ordered behavior, not a set: JSON parsing, structural checks,
|
||||||
and records chunk metadata for start unit, end unit, and unit count.
|
domain-specific checks, durable-schema checks, and advisory checks may have
|
||||||
|
different responsibilities and failure handling. The active default chains and
|
||||||
|
override rules are maintained in
|
||||||
|
[Configuration](../config.md#production-validator-keys-and-default-chains).
|
||||||
|
|
||||||
Options:
|
Reference slots are part of the module specification. They describe the
|
||||||
|
accepted artifact kind, media type, size, and whether a binding is required;
|
||||||
|
the framework validates those constraints before construction. An external
|
||||||
|
reference is materialized during preparation. A generated reference is a
|
||||||
|
compatible normalized artifact handed from an earlier pipeline step at
|
||||||
|
operation time. The configuration reference rules, including precedence and
|
||||||
|
ordered-handoff requirements, are maintained in
|
||||||
|
[Configuration](../config.md#references-and-ordered-handoffs).
|
||||||
|
|
||||||
- `max_units`: positive integer, default `50`;
|
## Focused Verification
|
||||||
- `overlap_units`: non-negative integer, default `0`, and less than
|
|
||||||
`max_units`.
|
|
||||||
|
|
||||||
Provides:
|
Exercise the leaf implementation and its registration path when changing a
|
||||||
|
module. Registry and registrar tests cover duplicate keys, required registries,
|
||||||
|
and typed construction; pipeline resolution tests cover capabilities, options,
|
||||||
|
and reference compatibility. Domain packages should additionally test their
|
||||||
|
codecs, validators, normalizers, and any integration handoffs they own.
|
||||||
|
|
||||||
- `chunks`
|
Run the affected package tests while iterating. The complete module suite is:
|
||||||
|
|
||||||
## `dnd/spells` Extractor
|
~~~sh
|
||||||
|
go test ./internal/modules/...
|
||||||
Package: `internal/modules/extract/dnd/spells`
|
~~~
|
||||||
|
|
||||||
The `dnd/spells` extractor owns D&D spell-cast artifact semantics. It renders
|
|
||||||
embedded prompts, loads the embedded structured response schema, calls the
|
|
||||||
structured LLM client, converts spell-cast responses into artifact candidates,
|
|
||||||
and supplies deterministic validators.
|
|
||||||
|
|
||||||
Requires:
|
|
||||||
|
|
||||||
- `chunks`
|
|
||||||
- `source.transcript`
|
|
||||||
|
|
||||||
Provides:
|
|
||||||
|
|
||||||
- `dnd.spell_casts`
|
|
||||||
|
|
||||||
Artifact type and schema version:
|
|
||||||
|
|
||||||
- artifact type: `dnd.spell_cast`
|
|
||||||
- schema version: `v1`
|
|
||||||
|
|
||||||
The extractor adds prompt and response-schema provenance to lane manifest
|
|
||||||
metadata. Durable artifact payload details belong in the
|
|
||||||
[D&D spell artifact contract](../integrations/dnd-spell-artifacts.md).
|
|
||||||
|
|
||||||
## D&D Spell Validators
|
|
||||||
|
|
||||||
The spell extractor returns two built-in validators:
|
|
||||||
|
|
||||||
- `dnd/spells/shape`: rejects malformed payloads and missing required fields.
|
|
||||||
- `dnd/spells/source_refs`: rejects candidates without valid source references.
|
|
||||||
|
|
||||||
Reason codes include:
|
|
||||||
|
|
||||||
- `invalid_payload`
|
|
||||||
- `missing_required_field`
|
|
||||||
- `missing_source_ref`
|
|
||||||
- `invalid_source_ref`
|
|
||||||
|
|
||||||
These validators are supplied by the extractor when no validators are configured
|
|
||||||
for the lane.
|
|
||||||
|
|
||||||
## `appendorder` Merger
|
|
||||||
|
|
||||||
Package: `internal/modules/merge/appendorder`
|
|
||||||
|
|
||||||
The `appendorder` merger clones and appends candidates in chunk order. It does
|
|
||||||
not deduplicate or reconcile candidates.
|
|
||||||
|
|
||||||
Provides:
|
|
||||||
|
|
||||||
- `merged`
|
|
||||||
|
|
||||||
## `noop` Normalizer
|
|
||||||
|
|
||||||
Package: `internal/modules/normalize/noop`
|
|
||||||
|
|
||||||
The `noop` normalizer clones merged candidates and returns them unchanged.
|
|
||||||
|
|
||||||
Requires:
|
|
||||||
|
|
||||||
- `merged`
|
|
||||||
|
|
||||||
Provides:
|
|
||||||
|
|
||||||
- `normalized`
|
|
||||||
|
|
||||||
## `json` Output
|
|
||||||
|
|
||||||
Package: `internal/modules/output/json`
|
|
||||||
|
|
||||||
The `json` output encoder converts approved artifacts, rejected artifacts,
|
|
||||||
warnings, and the run manifest into logical JSON output files. It groups
|
|
||||||
approved artifacts by artifact type and sanitizes artifact-type file names.
|
|
||||||
|
|
||||||
Requires:
|
|
||||||
|
|
||||||
- `normalized`
|
|
||||||
|
|
||||||
Provides:
|
|
||||||
|
|
||||||
- `encoded`
|
|
||||||
|
|
||||||
Durable output file shapes belong in the
|
|
||||||
[JSON output contract](../integrations/json-output.md). Operator behavior
|
|
||||||
belongs in [Operations](../operations.md).
|
|
||||||
|
|
||||||
## Production Registration
|
|
||||||
|
|
||||||
Production registration is centralized in `internal/cli/catalog.go`.
|
|
||||||
|
|
||||||
Do not make framework code import production modules. The CLI wires production
|
|
||||||
modules at the application boundary; tests may provide fake registries or fake
|
|
||||||
catalogs directly.
|
|
||||||
|
|
||||||
## Adding A Module
|
|
||||||
|
|
||||||
When adding a module, keep source-format and extraction-domain boundaries clear:
|
|
||||||
|
|
||||||
- input modules may know external source formats;
|
|
||||||
- extract modules may know artifact semantics and prompt/schema assets;
|
|
||||||
- merge and normalize modules own candidate combination and reconciliation;
|
|
||||||
- output modules own serialization, not diagnostics or CLI reporting.
|
|
||||||
|
|
||||||
Update [Development](../policy/development.md), [Configuration](../config.md),
|
|
||||||
internal docs, integration docs, and examples when the new module becomes
|
|
||||||
implemented production behavior.
|
|
||||||
|
|||||||
@@ -1,86 +1,58 @@
|
|||||||
# Internal Overview
|
# Internal Overview
|
||||||
|
|
||||||
This directory documents implemented Notarius internals for developers and LLM
|
This document is the implemented component map for Notarius. Normative
|
||||||
coding agents. It complements [Architecture](../policy/architecture.md), which
|
boundaries and dependency direction belong in
|
||||||
is the durable policy for boundaries and invariants.
|
[Architecture](../policy/architecture.md). User and operator contracts belong
|
||||||
|
in the [CLI](../cli.md), [Configuration](../config.md),
|
||||||
|
[Operations](../operations.md), and [integration contracts](../integrations/).
|
||||||
|
|
||||||
## Executable And CLI
|
## Execution Path
|
||||||
|
|
||||||
`cmd/notarius` calls the CLI package. `internal/cli` owns:
|
~~~
|
||||||
|
cmd/notarius -> internal/cli -> configuration and production composition
|
||||||
|
-> internal/framework/pipeline -> logical output files
|
||||||
|
-> internal/cli -> durable output and optional state/debug data
|
||||||
|
~~~
|
||||||
|
|
||||||
- command parsing and usage;
|
The CLI is the application boundary: it discovers configuration, composes
|
||||||
- config discovery and loading;
|
production registries and runtime collaborators, invokes the framework, and
|
||||||
- production module catalog and registry wiring;
|
places returned files. The framework resolves and prepares a fixed extraction
|
||||||
- production LLM client construction;
|
pipeline, then returns logical results without owning process behavior or
|
||||||
- run directory creation;
|
physical state roots.
|
||||||
- durable output writes;
|
|
||||||
- user-facing stdout, stderr, and exit codes.
|
|
||||||
|
|
||||||
The CLI should stay thin around framework contracts. Domain extraction behavior
|
## Components
|
||||||
belongs in modules, not in command handlers.
|
|
||||||
|
|
||||||
## Core Packages
|
| Area | Implemented owners | Responsibility |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| Executable and command boundary | **cmd/notarius**, **internal/cli** | Process entry, command dispatch, configuration discovery, production composition, runtime collaborator setup, durable file placement, and user-facing reporting. |
|
||||||
|
| Configuration | **internal/core/config** | Defaults, strict YAML parsing, environment overrides, structural validation, effective resolution, redaction, and resolved-composition summaries. |
|
||||||
|
| Generic models | **internal/core/source**, **internal/core/artifacts**, **internal/framework/contracts** | Source documents and chunks, manifests and provenance, plus typed artifact, reference, validation, output, and structured-completion contracts. |
|
||||||
|
| Pipeline framework | **internal/framework/pipeline** | Registries, profile and reference resolution, typed preparation, validation, retry coordination, ordered execution, handoff, and result assembly. |
|
||||||
|
| LLM and prompt runtime | **internal/framework/llm**, **internal/framework/promptfs** | Provider-neutral structured completions, scheduling, profile recording, prompt assets, schema registration, and credential-shaped-value redaction. |
|
||||||
|
| Runtime state | **internal/core/fileio**, **internal/core/debugbundle**, **internal/framework/checkpoint**, **internal/framework/chunkplan**, **internal/framework/chunkmap**, **internal/framework/debug** | Confined atomic files, debug bundles, checkpoint and chunk-plan state, accepted chunk maps, and pipeline-facing debug recording. |
|
||||||
|
| Production extensions | **internal/modules/generic**, **internal/modules/seriatim**, **internal/modules/dnd** | Domain-neutral extensions, Seriatim input support, and D&D extraction families registered into the production catalog. |
|
||||||
|
|
||||||
- `internal/core/artifacts`: artifact candidates, approved artifacts, rejected
|
Generic core and framework packages do not depend on production extensions.
|
||||||
artifacts, validation decisions, and run manifests.
|
Concrete extensions depend inward on their contracts and are registered only at
|
||||||
- `internal/core/config`: defaults, YAML config parsing, environment overrides,
|
the CLI composition boundary.
|
||||||
validation, redaction, and resolved pipeline config.
|
|
||||||
- `internal/core/diagnostics`: per-run diagnostics directory creation,
|
|
||||||
diagnostics artifact writers, atomic writes, and retention decisions.
|
|
||||||
- `internal/core/source`: source documents, source units, source references, and
|
|
||||||
validation.
|
|
||||||
|
|
||||||
Core packages should remain deterministic and concrete. They should not import
|
## Focused Documentation
|
||||||
production modules.
|
|
||||||
|
|
||||||
## Framework Packages
|
- [Configuration Internals](configuration.md): loading, validation, effective
|
||||||
|
resolution, redaction, and resolved-composition identity.
|
||||||
|
- [CLI Internals](cli.md): command dispatch, production composition, run
|
||||||
|
orchestration, and terminal reporting.
|
||||||
|
- [Pipeline Internals](pipeline.md): resolution, preparation, execution,
|
||||||
|
validation, typed handoff, and framework state hooks.
|
||||||
|
- [Run State Internals](state.md): output, cache, debug collaborator
|
||||||
|
composition, and path safety.
|
||||||
|
- [LLM Runtime](llm.md): structured completion, scheduling, prompt assets,
|
||||||
|
profiles, and secret handling.
|
||||||
|
- [Module Internals](modules.md): generic extension registration, module
|
||||||
|
construction, validation, and reference mechanics.
|
||||||
|
- [D&D Module Internals](dnd.md): shared D&D extractor conventions, generated
|
||||||
|
reference projections, and lane-specific exceptions. Durable D&D and
|
||||||
|
Seriatim data shapes remain in the [integration contracts](../integrations/).
|
||||||
|
|
||||||
- `internal/framework/contracts`: interfaces and request/result structs for
|
Use this map to find an owner, then read the focused document and its tests
|
||||||
input adapters, chunkers, extractors, mergers, normalizers, validators, output
|
before changing behavior.
|
||||||
encoders, and structured LLM clients.
|
|
||||||
- `internal/framework/pipeline`: module registries, module specs, profile
|
|
||||||
resolution, capability checks, run orchestration, warnings, validation, and
|
|
||||||
manifest population.
|
|
||||||
- `internal/framework/llm`: OpenAI-compatible structured-output client,
|
|
||||||
scheduler, schema registry, retries, and secret redaction.
|
|
||||||
- `internal/framework/prompt`: embedded prompt registry and template rendering.
|
|
||||||
- `internal/framework/validate`: validator decision helpers and cardinality
|
|
||||||
enforcement.
|
|
||||||
|
|
||||||
Framework code should stay source-agnostic and domain-agnostic.
|
|
||||||
|
|
||||||
## Module Packages
|
|
||||||
|
|
||||||
Production module packages live under `internal/modules`:
|
|
||||||
|
|
||||||
- `input/seriatim`
|
|
||||||
- `chunk/generic`
|
|
||||||
- `extract/dnd/spells`
|
|
||||||
- `merge/appendorder`
|
|
||||||
- `normalize/noop`
|
|
||||||
- `output/json`
|
|
||||||
|
|
||||||
Each module package owns its contract implementation, module spec,
|
|
||||||
registration, options, focused tests, and module-specific errors.
|
|
||||||
|
|
||||||
## Fixtures And Tests
|
|
||||||
|
|
||||||
The repository uses focused package tests plus a fixture-driven CLI workflow.
|
|
||||||
|
|
||||||
- CLI acceptance tests cover maintained examples under `examples/`.
|
|
||||||
- Pipeline tests cover registry composition and end-to-end framework behavior
|
|
||||||
with fakes.
|
|
||||||
- Module tests cover implemented module contracts without requiring real
|
|
||||||
provider calls.
|
|
||||||
- LLM tests use local test servers and fakes.
|
|
||||||
|
|
||||||
Do not use real external services in tests. Use fakes, fixtures, or local test
|
|
||||||
servers.
|
|
||||||
|
|
||||||
## Boundary Reminders
|
|
||||||
|
|
||||||
- Source-format details stay in input modules and integration docs.
|
|
||||||
- Extraction-domain details stay in extract modules and artifact docs.
|
|
||||||
- Provider wire details stay in the LLM runtime and provider integration docs.
|
|
||||||
- Durable output contracts belong in integration docs.
|
|
||||||
- Operator procedures belong in `docs/operations.md`, not internal docs.
|
|
||||||
|
|||||||
@@ -1,127 +1,180 @@
|
|||||||
# Pipeline Internals
|
# Pipeline Internals
|
||||||
|
|
||||||
The implemented pipeline runner lives in `internal/framework/pipeline`. It
|
This document describes the framework-owned pipeline mechanics in
|
||||||
executes the fixed workflow defined by the architecture policy:
|
**internal/framework/pipeline**. [Configuration](../config.md) owns selectable
|
||||||
|
profiles, bindings, and retry settings; [Operations](../operations.md) owns
|
||||||
|
state lifecycle and recovery; and the [integration contracts](../integrations/)
|
||||||
|
own durable output shapes. Concrete production extensions are covered by
|
||||||
|
[Module Internals](modules.md).
|
||||||
|
|
||||||
```text
|
## Boundary
|
||||||
|
|
||||||
|
The pipeline framework accepts a resolved composition, registries, shared
|
||||||
|
dependencies, input bytes, and state/debug collaborators. It returns logical
|
||||||
|
output files, normalized artifacts, recorded rejections and warnings, manifest
|
||||||
|
provenance, and checkpoint decisions. The CLI owns process arguments,
|
||||||
|
configuration discovery, physical roots, and placement of returned output
|
||||||
|
files.
|
||||||
|
|
||||||
|
The framework has one fixed shape:
|
||||||
|
|
||||||
|
~~~
|
||||||
input -> chunk -> extract -> merge -> normalize -> output
|
input -> chunk -> extract -> merge -> normalize -> output
|
||||||
```
|
~~~
|
||||||
|
|
||||||
Pipeline execution is serial. The runner executes the resolved lanes one after
|
Input and chunking are pipeline-wide. A selected artifact lane owns extract,
|
||||||
another in the fixed workflow order.
|
merge, and normalize; output aggregates the terminal lane outcomes. A pipeline
|
||||||
|
is an ordered list of steps, not an arbitrary workflow graph.
|
||||||
|
|
||||||
## Profile Resolution
|
## Resolve, Materialize, Prepare
|
||||||
|
|
||||||
Config loading produces `pipeline.PipelineProfile` values. Resolution happens
|
Resolution turns a configured pipeline profile into a **ResolvedPipeline**.
|
||||||
before execution:
|
It normalizes the pipeline and lane identities, applies stage defaults, selects
|
||||||
|
requested lanes where that is supported, resolves validator chains, checks
|
||||||
|
module capabilities and typed artifact compatibility, validates options, and
|
||||||
|
assigns a deterministic resolved-composition digest. The resolved pipeline
|
||||||
|
contains bindings and declared reference targets, not external reference bytes.
|
||||||
|
Configuration resolution supplies the selected profile and catalog; see
|
||||||
|
[Configuration Internals](configuration.md).
|
||||||
|
|
||||||
1. `internal/core/config.Config.Resolve` validates config and finds the named
|
External reference materialization happens before preparation. The materializer
|
||||||
pipeline.
|
checks that each slot is declared by the selected module, resolves a file path
|
||||||
2. The optional lane selection is passed to `pipeline.ResolvePipeline`.
|
relative to the correct configuration or working-directory origin, reads
|
||||||
3. Module bindings are defaulted:
|
UTF-8 text, verifies media type and size limits, and retains bounded
|
||||||
- chunk: `generic`
|
provenance. A generated-artifact selector remains declared but has no bytes
|
||||||
- merge: `appendorder`
|
until its producing step completes.
|
||||||
- normalize: `noop`
|
|
||||||
- output: `json`
|
|
||||||
- LLM profile: `default`
|
|
||||||
4. The module catalog is checked for each bound module key.
|
|
||||||
5. Module capabilities are checked in workflow order.
|
|
||||||
6. A digest is calculated from the resolved pipeline without the digest field.
|
|
||||||
|
|
||||||
The CLI writes the resolved pipeline and digest to diagnostics.
|
Preparation is the construction boundary. It validates the resolved shape and
|
||||||
|
registry set, clones the resolved data, then constructs the input adapter,
|
||||||
|
chunker, stage-local validators, every typed lane, and output encoder with
|
||||||
|
cloned options, references, and shared dependencies. It also collects stable
|
||||||
|
checkpoint fingerprints. Missing registrations, incompatible typed entries,
|
||||||
|
nil implementations, and constructor failures are reported before source
|
||||||
|
parsing or any stage operation begins.
|
||||||
|
|
||||||
## Registries And Module Specs
|
An output encoder can opt into source-evidence publication through its output
|
||||||
|
policy. Preparation keeps the configured lane allowlist and active lanes
|
||||||
|
separate, then verifies an exact typed evidence projector and registered codec
|
||||||
|
for each active lane. The resulting private plan is immutable; lanes excluded
|
||||||
|
by invocation filtering remain configured but do not acquire a projector for
|
||||||
|
that run.
|
||||||
|
|
||||||
`pipeline.Registries` holds concrete constructors for execution. A
|
## Typed Lanes And References
|
||||||
`pipeline.ModuleCatalog` exposes module specs for config validation and
|
|
||||||
resolution.
|
|
||||||
|
|
||||||
Every production module registers a `ModuleSpec` with:
|
Each resolved lane has one artifact kind, codec, and exact Go type. The
|
||||||
|
framework uses private type erasure only around those typed operations; every
|
||||||
|
handoff checks exact type and codec identity and reports incompatibility as an
|
||||||
|
error rather than panicking. Encoding through the registered codec is the
|
||||||
|
boundary for output, checkpoints, debug records, and generated references.
|
||||||
|
|
||||||
- `Key`: module key used in config;
|
Reference targets are stage- and lane-specific. External reference bytes are
|
||||||
- `Stage`: module kind such as input, chunk, extract, merge, normalize,
|
cloned into the operation request. Generated references are built at the next
|
||||||
validate, or output;
|
step boundary from exactly one accepted normalized producer output. The
|
||||||
- `Provides`: capabilities added after that module runs;
|
framework decodes and re-encodes that output with the registered producer
|
||||||
- `Requires`: capabilities that must already be available.
|
codec, checks its complete schema and media identity, and records a content
|
||||||
|
digest plus bounded producer provenance. A missing, ambiguous, invalid, or
|
||||||
|
incompatible producer prevents the consumer step from starting.
|
||||||
|
|
||||||
Capability checks prevent incompatible pipeline composition before a run starts.
|
## Execution And Ordering
|
||||||
|
|
||||||
## Runner Input And Output
|
The runner validates its input, installs no-op state collaborators when none
|
||||||
|
were supplied, and serially performs source parsing and chunk-plan selection.
|
||||||
|
An accepted plan is materialized into source-addressed chunks and passes the
|
||||||
|
configured chunk validators before any lane runs. A chunk rejection is a
|
||||||
|
recorded pipeline outcome: lanes do not start, but the output stage can encode
|
||||||
|
the terminal result.
|
||||||
|
|
||||||
`pipeline.RunInput` carries:
|
For each ordered step, the runner first builds generated reference sets from
|
||||||
|
the accepted normalized outputs of earlier steps. It then executes the step's
|
||||||
|
lanes. Later steps do not begin until the current step is terminal and its
|
||||||
|
generated handoffs have succeeded.
|
||||||
|
|
||||||
- a `ResolvedPipeline`;
|
Within a step, the lane engine dispatches extraction jobs in deterministic
|
||||||
- optional source ID, input path, and raw input bytes;
|
chunk-first, lane-second order to a bounded worker group. When all extraction
|
||||||
- a structured LLM client;
|
jobs for one lane are terminal, a bounded continuation group can run that
|
||||||
- run ID, start time, LLM profile manifest metadata, and CLI metadata.
|
lane's merge and normalize work while extraction for other lanes continues.
|
||||||
|
The framework does not create an unbounded goroutine per chunk or lane.
|
||||||
|
|
||||||
`pipeline.RunOutput` carries:
|
Completion timing does not determine public results. The coordinator restores
|
||||||
|
lane and chunk order before merging results, and selects a framework error by
|
||||||
|
stable stage, lane, and chunk position. A validator rejection records a lane
|
||||||
|
outcome without cancelling unrelated work. A framework error or parent
|
||||||
|
cancellation cancels derived work, prevents queued work from starting, waits
|
||||||
|
for started workers, and prevents output encoding.
|
||||||
|
|
||||||
- run manifest;
|
## Validation, Retries, And Output
|
||||||
- approved artifacts;
|
|
||||||
- rejected artifacts;
|
|
||||||
- warnings;
|
|
||||||
- logical output files returned by the output encoder.
|
|
||||||
|
|
||||||
The CLI owns durable file writes and diagnostics writes after the runner returns.
|
Every chunk, extract, merge, and normalize candidate passes its resolved
|
||||||
|
validator chain. Validators receive immutable canonical input appropriate to
|
||||||
|
their target: chunks, typed values, or serialized codec bytes. They may
|
||||||
|
approve, approve with warnings, reject, or fail. A rejection is an ordinary
|
||||||
|
pipeline result; a validator error is a framework error.
|
||||||
|
|
||||||
## Execution
|
The runner applies the binding's retry policy around a stage operation and its
|
||||||
|
complete validation chain. It preserves warnings only from the final accepted
|
||||||
|
or rejected attempt. Cancellation stops retries. Normalizer-specific retry
|
||||||
|
directives consume this same budget and validate any final safe fallback through
|
||||||
|
the normalizer chain.
|
||||||
|
|
||||||
The runner:
|
After terminal lane work, the runner assembles manifest provenance, normalized
|
||||||
|
artifacts, rejections, warnings, and an optional accepted chunk map. When an
|
||||||
|
output policy selected evidence lanes, it decodes accepted serialized normalize
|
||||||
|
outputs through their registered codecs and invokes the prepared typed
|
||||||
|
projectors. Rejected or absent lanes contribute nothing. This reconstruction is
|
||||||
|
also used after normalized-checkpoint reuse, so no second typed output channel
|
||||||
|
is retained. The runner passes the resulting owned artifact to the output
|
||||||
|
encoder, which returns logical files and does not choose a physical directory.
|
||||||
|
The CLI publishes those files only after the runner returns without a framework
|
||||||
|
error. Logical file names and schemas are defined by the [output integration
|
||||||
|
contracts](../integrations/).
|
||||||
|
|
||||||
1. validates run input and registries;
|
## Checkpoint And Debug Hooks
|
||||||
2. builds the input adapter and parses the raw input into a source document;
|
|
||||||
3. validates the source document;
|
|
||||||
4. builds the chunker and produces source chunks;
|
|
||||||
5. runs each selected artifact lane in sorted resolved order;
|
|
||||||
6. builds the output encoder and validates logical output file names.
|
|
||||||
|
|
||||||
Within an artifact lane, the runner:
|
The runner receives checkpoint and debug interfaces rather than roots. It
|
||||||
|
records workflow transitions and reuse decisions through the supplied
|
||||||
|
collaborators, and clones reusable artifacts before they re-enter normal typed
|
||||||
|
handoff. Generated-reference dependencies participate in checkpoint decisions.
|
||||||
|
Selective recomputation can require a canonical accepted normalized predecessor
|
||||||
|
before a dependent lane starts.
|
||||||
|
|
||||||
1. builds the extractor, merger, and normalizer;
|
Debug recording is attempt-scoped and application-owned. A failure to persist
|
||||||
2. records module manifest metadata when modules provide it;
|
required debug data is a framework error. State roots, persistence, reason-code
|
||||||
3. extracts candidates from each chunk;
|
meanings, resume, and cleanup are intentionally owned by
|
||||||
4. normalizes candidate envelope fields such as index, extractor key, artifact
|
[Run State Internals](state.md) and [Operations](../operations.md).
|
||||||
type, and schema version;
|
|
||||||
5. merges candidates;
|
|
||||||
6. normalizes merged candidates;
|
|
||||||
7. validates candidate envelope consistency;
|
|
||||||
8. runs validators;
|
|
||||||
9. converts approved candidates to artifacts.
|
|
||||||
|
|
||||||
## Validators
|
## Invariants To Preserve
|
||||||
|
|
||||||
If a lane declares validators in config, the runner builds those validators from
|
- The six fixed stages remain explicit; a pipeline is not a general DAG.
|
||||||
the validator registry. Otherwise it uses validators returned by the extractor.
|
- Resolution and preparation reject statically discoverable incompatibility
|
||||||
|
before parsing or execution.
|
||||||
|
- Every typed lane uses one compatible artifact kind, codec, and exact Go type.
|
||||||
|
- Generated references come only from one earlier accepted normalized producer
|
||||||
|
and carry canonical identity rather than an unverified value.
|
||||||
|
- Rejections are recorded outcomes; framework errors cancel derived work and
|
||||||
|
prevent output encoding.
|
||||||
|
- Public ordering and selected errors are independent of goroutine completion
|
||||||
|
order.
|
||||||
|
- Pipeline modules receive collaborators and data, never CLI streams or
|
||||||
|
physical output, cache, or debug roots.
|
||||||
|
|
||||||
Each validator must return exactly one decision for each eligible candidate. The
|
## Focused Tests
|
||||||
runner enforces decision cardinality with `internal/framework/validate`.
|
|
||||||
Rejected candidates are removed before the next validator runs. Approved
|
|
||||||
candidates continue through the chain.
|
|
||||||
|
|
||||||
The production CLI currently registers no standalone validator modules. The
|
- **internal/framework/pipeline/profile_test.go** and
|
||||||
current D&D spell extractor supplies deterministic shape and source-reference
|
**typed_resolution_test.go** cover resolution, defaults, ordered steps,
|
||||||
validators.
|
compatibility, validators, references, and resolved identity.
|
||||||
|
- **internal/framework/pipeline/preparation_test.go** covers complete
|
||||||
|
construction before execution and contextual construction failures.
|
||||||
|
- **internal/framework/pipeline/references_test.go** and **handoff_test.go**
|
||||||
|
cover external materialization, generated references, provenance, and typed
|
||||||
|
producer checks.
|
||||||
|
- **internal/framework/pipeline/runner_concurrency_test.go** covers bounded
|
||||||
|
execution, ordered steps, stable error selection, rejections, and
|
||||||
|
cancellation.
|
||||||
|
- **internal/framework/pipeline/runner_chunk_plan_test.go**,
|
||||||
|
**runner_typed_checkpoint_test.go**, and
|
||||||
|
**runner_accepted_checkpoint_test.go** cover state hooks and reuse behavior.
|
||||||
|
- **internal/framework/pipeline/runner_attempt_debug_test.go** and
|
||||||
|
**runner_terminal_debug_test.go** cover attempt and terminal debug behavior.
|
||||||
|
|
||||||
## Warnings And Failures
|
Run **go test ./internal/framework/pipeline ./internal/cli** after changing a
|
||||||
|
pipeline boundary. Use the more focused tests above while iterating.
|
||||||
Warnings from chunking, extraction, merging, normalization, validation, and
|
|
||||||
output encoding are accumulated in `RunOutput.Warnings`.
|
|
||||||
|
|
||||||
Errors wrap the operation and module key or lane context. If execution fails
|
|
||||||
after a manifest exists, the returned manifest is marked `failed` and receives a
|
|
||||||
completion timestamp.
|
|
||||||
|
|
||||||
On successful execution, the manifest validation status is:
|
|
||||||
|
|
||||||
- `approved` when no candidates were rejected;
|
|
||||||
- `rejected` when at least one candidate was rejected.
|
|
||||||
|
|
||||||
## Manifest Population
|
|
||||||
|
|
||||||
The manifest records run ID, pipeline ID, pipeline digest, module keys, artifact
|
|
||||||
lanes, LLM profile metadata, source digest, validation status, and timing.
|
|
||||||
|
|
||||||
Modules can add non-secret manifest metadata by implementing
|
|
||||||
`contracts.ManifestMetadataProvider`. The D&D spell extractor uses this for
|
|
||||||
prompt and response-schema provenance.
|
|
||||||
|
|||||||
145
docs/internal/state.md
Normal file
145
docs/internal/state.md
Normal file
@@ -0,0 +1,145 @@
|
|||||||
|
# Run State Internals
|
||||||
|
|
||||||
|
This document describes the implementation collaborators behind output, cache,
|
||||||
|
and debug state. User-visible fields belong in [Configuration](../config.md),
|
||||||
|
and physical layout, retention, recovery, reason codes, and cleanup belong in
|
||||||
|
[Operations](../operations.md).
|
||||||
|
|
||||||
|
## Composition
|
||||||
|
|
||||||
|
`internal/cli` is the only physical-path composition root. It resolves the
|
||||||
|
effective configuration, selects exact roots, allocates requested debug bundles,
|
||||||
|
constructs cache collaborators, writes logical output files, and reports paths.
|
||||||
|
Pipeline modules receive interfaces and request data, never output, cache, or
|
||||||
|
debug roots.
|
||||||
|
|
||||||
|
The CLI creates no chunk-plan store in bypass mode. It creates a checkpoint
|
||||||
|
recorder only when recording is enabled and a checkpoint loader only for a
|
||||||
|
resume invocation. It allocates debug state only after a safe run identity has
|
||||||
|
been generated and only when debug capture was requested. These choices keep
|
||||||
|
the three state families independently composable.
|
||||||
|
|
||||||
|
## Output And Cache
|
||||||
|
|
||||||
|
The pipeline runner returns logical output files. After validating every
|
||||||
|
logical name, the CLI exclusively creates the run directory beneath the
|
||||||
|
selected output root and performs confined, atomic file writes within it.
|
||||||
|
The runner supplies an accepted chunk map as an optional, defensively owned
|
||||||
|
output-request artifact. The JSON encoder alone decides whether its explicit
|
||||||
|
option writes the map and optional index descriptor; neither the map payload
|
||||||
|
nor its annotations are copied into the run manifest. The durable fields are
|
||||||
|
owned by the [Accepted Chunk Map contract](../integrations/chunk-map.md).
|
||||||
|
|
||||||
|
`internal/framework/chunkplan` owns source-addressed plan storage, validation,
|
||||||
|
and atomic publication. Its store is constructed only when the selected mode is
|
||||||
|
not `bypass`.
|
||||||
|
|
||||||
|
`internal/framework/checkpoint` owns checkpoint identity, manifests, payload
|
||||||
|
codecs, loader, and recorder. The CLI constructs a recorder whenever checkpoint
|
||||||
|
recording is enabled and constructs a loader only for a `--resume` invocation.
|
||||||
|
Identity incorporates explicit stable semantic fingerprints collected from
|
||||||
|
prepared modules and validators in addition to configuration, input,
|
||||||
|
references, runtime overrides, and LLM profiles.
|
||||||
|
The serialized
|
||||||
|
`workspace_schema_version` identifiers are frozen wire-compatibility fields;
|
||||||
|
they do not describe a current public state surface.
|
||||||
|
|
||||||
|
Ordered-step lane checkpoints include the step identity in their storage scope.
|
||||||
|
When a later lane consumes a generated artifact, its dependency fingerprints
|
||||||
|
include the producer's artifact kind, complete schema identity, media type,
|
||||||
|
canonical content digest, and size. Ordinary resume compares those fingerprints
|
||||||
|
when progressively loading consumer stage checkpoints, so changed producer
|
||||||
|
content produces `dependency_invalidated` rather than stale downstream reuse.
|
||||||
|
Selective recomputation instead requires each unselected producer's accepted
|
||||||
|
normalized artifact; invalid accepted state records its specific bounded reason
|
||||||
|
and stops before the dependent. The selected step and its transitive dependents
|
||||||
|
record `forced_recompute`.
|
||||||
|
|
||||||
|
Ordinary resume loads extract, merge, and normalize checkpoints progressively
|
||||||
|
and may execute later lane stages after an earlier cache miss. Selective
|
||||||
|
recomputation instead asks the loader for the required producer's accepted
|
||||||
|
normalize artifact. That lookup reuses the existing normalize files, requires
|
||||||
|
workspace schema v3 plus an exact non-empty invocation identity, and deliberately
|
||||||
|
does not require extract or merge checkpoint files or dependency fingerprints.
|
||||||
|
The runner performs canonical codec and producer-provenance validation before
|
||||||
|
cloning the artifact into normal step output. Success restores only stored
|
||||||
|
normalize warnings and emits one normalize decision; failure retains the files,
|
||||||
|
records the decision, and stops without executing the producer or consumer.
|
||||||
|
|
||||||
|
The loader assigns a typed category and reason code at each validation site;
|
||||||
|
diagnostic prose is not classified after the fact. The runner then applies
|
||||||
|
forced-execution policy, validates reusable artifact bytes through the prepared
|
||||||
|
codec once, returns the canonical hydrated value to the stage, and records the
|
||||||
|
final decision before enforcing a required-predecessor failure. That failure
|
||||||
|
names only the step, lane, and stable reason code. Decision detail is selected
|
||||||
|
from code-owned descriptions by reason code and then UTF-8 normalized and
|
||||||
|
bounded; callers cannot supply arbitrary diagnostic prose. Typed categories and
|
||||||
|
codes remain intact through pipeline events and become strings only in manifest
|
||||||
|
and debug-summary JSON.
|
||||||
|
[Operations](../operations.md#checkpoint-recording-resume-and-recompute) is the
|
||||||
|
canonical operator-facing reason-code reference.
|
||||||
|
|
||||||
|
`internal/core/fileio` provides confined atomic file writes used by state
|
||||||
|
collaborators. The chunk-plan store retains its stronger entry validation.
|
||||||
|
|
||||||
|
The CLI constructs selective-recomputation policy from resolved generated
|
||||||
|
artifact dependencies. It forces the selected step and transitive consumers,
|
||||||
|
while marking unforced producers as required reusable inputs. The runner owns
|
||||||
|
the actual hydration and rejection decisions; the [Operations guide](../operations.md#checkpoint-recording-resume-and-recompute)
|
||||||
|
owns the operator workflow and stable reason-code meanings.
|
||||||
|
|
||||||
|
## Debug Bundles
|
||||||
|
|
||||||
|
`internal/core/debugbundle` allocates an explicitly requested per-run bundle
|
||||||
|
with `summary/` and `trace/` roots. `SummaryWriter` persists redacted command,
|
||||||
|
resolution, run, warning, and failure artifacts. `internal/framework/debug`
|
||||||
|
implements the pipeline-facing trace recorder under the trace root.
|
||||||
|
|
||||||
|
The CLI allocates a bundle before pipeline resolution and treats requested
|
||||||
|
summary or trace persistence failures as command failures. The pipeline's debug
|
||||||
|
boundaries redact sensitive metadata and credential-shaped bytes while allowing
|
||||||
|
application-owned trace material. Debug data is never a checkpoint source or
|
||||||
|
cache input.
|
||||||
|
|
||||||
|
Generated reference bytes exist only in cloned operation requests and are not
|
||||||
|
written as paths into checkpoints, manifests, or debug summaries. Those state
|
||||||
|
surfaces retain canonical identities and bounded producer provenance so that a
|
||||||
|
resume decision can be explained without copying generated campaign content.
|
||||||
|
|
||||||
|
After allocation, one CLI-owned state value accumulates the known report paths,
|
||||||
|
pipeline outcome counts, and validation status. A single guarded terminalization
|
||||||
|
operation writes the success report, or makes one attempt each to write the
|
||||||
|
failure report and error log. Terminal persistence failures are reported
|
||||||
|
separately and never replace the command's primary error.
|
||||||
|
|
||||||
|
## Invariants To Preserve
|
||||||
|
|
||||||
|
- Modules receive state collaborators and request data, never physical roots.
|
||||||
|
- Output logical paths are validated before a run directory is allocated, and
|
||||||
|
files are atomically written within that directory.
|
||||||
|
- Chunk-plan publication occurs only for accepted plans; bypass does not
|
||||||
|
construct or touch a plan store.
|
||||||
|
- Checkpoint recording and checkpoint loading remain separate collaborators.
|
||||||
|
- Debug state is opt-in, is not cache input, and terminal reporting does not
|
||||||
|
obscure the command's primary failure.
|
||||||
|
|
||||||
|
## Tests To Inspect
|
||||||
|
|
||||||
|
- `internal/cli/run_contract_test.go`: command-owned state allocation,
|
||||||
|
terminalization, and output/report boundaries.
|
||||||
|
- `internal/cli/cache_contract_test.go`: cache-mode precedence, root selection,
|
||||||
|
and resume collaborator construction.
|
||||||
|
- `internal/cli/state_hardening_test.go`: independent roots, reuse, failures,
|
||||||
|
permissions, cleanup, and redaction.
|
||||||
|
- `internal/cli/recompute_policy_test.go`: forced dependents and required
|
||||||
|
reusable predecessors for selective recomputation.
|
||||||
|
- `internal/cli/recompute_execution_contract_test.go`: selective recomputation,
|
||||||
|
filesystem recovery, deterministic decisions, and failed predecessor state.
|
||||||
|
- `internal/cli/production_contract_test.go`: production composition and
|
||||||
|
configuration validation at the CLI boundary.
|
||||||
|
- `internal/cli/example_contract_test.go`: maintained example ownership.
|
||||||
|
- `internal/core/debugbundle/*_test.go`: bundle allocation and summary writes.
|
||||||
|
- `internal/framework/checkpoint/*_test.go`: checkpoint serialization and
|
||||||
|
reuse.
|
||||||
|
- `internal/framework/chunkplan/store_test.go`: plan envelope, confinement,
|
||||||
|
publication, and permissions.
|
||||||
@@ -1,132 +1,216 @@
|
|||||||
# Operations
|
# Operations
|
||||||
|
|
||||||
This is the canonical reference for operating implemented Notarius runs.
|
This is the canonical guide for operating Notarius runtime state. The
|
||||||
|
[CLI reference](cli.md) owns command syntax and exit statuses, while
|
||||||
|
[Configuration](config.md) owns fields, defaults, and precedence. Maintainers
|
||||||
|
who need implementation mechanics should read [Run State Internals](internal/state.md).
|
||||||
|
|
||||||
## Normal Run
|
## State Surfaces
|
||||||
|
|
||||||
A run reads one source file, resolves one configured pipeline, calls the
|
Each run can use independent roots with different retention and access-control
|
||||||
configured OpenAI-compatible LLM profile, writes durable JSON output, and writes
|
needs.
|
||||||
diagnostics for inspection.
|
|
||||||
|
|
||||||
```sh
|
| Surface | Purpose | Created when | Retention |
|
||||||
go run ./cmd/notarius run dnd-session \
|
| --- | --- | --- | --- |
|
||||||
--config examples/dnd-spells.config.yml \
|
| Output | Durable user-facing result bundle | A pipeline completes and returns logical output files | Keep until consumers no longer need it. |
|
||||||
--input examples/seriatim-minimal-transcript.json \
|
| Chunk-plan cache | Reconstructible source-addressed plan | The configured cache mode permits cache I/O | Keep while reuse is useful. |
|
||||||
--output-dir ./notarius-output \
|
| Checkpoint cache | Reconstructible execution and recovery state | Checkpoint recording is enabled | Keep only while recovery or reuse is useful. |
|
||||||
--diagnostics-dir /tmp/notarius
|
| Debug bundle | Explicit diagnostic record | A run requests debug collection | Keep only under an intentional sensitive-data retention policy. |
|
||||||
```
|
|
||||||
|
|
||||||
The command prints a success line with the pipeline ID, approved and rejected
|
Output, cache, and debug roots are never merged or cleaned automatically. Use
|
||||||
artifact counts, and the output path.
|
separate locations and permissions for operators or services that must not
|
||||||
|
share application data.
|
||||||
|
|
||||||
## Output Directory
|
## Roots And Permissions
|
||||||
|
|
||||||
Durable output is written to:
|
The configured output and debug directories are exact roots. An empty cache
|
||||||
|
directory selects a per-user root:
|
||||||
|
|
||||||
```text
|
~~~
|
||||||
|
<os.UserCacheDir>/notarius/chunk-plans
|
||||||
|
<os.UserCacheDir>/notarius/checkpoints
|
||||||
|
~~~
|
||||||
|
|
||||||
|
The field definitions and configuration examples are in [Configuration](config.md).
|
||||||
|
On supported Unix systems, output directories and files are created with
|
||||||
|
requested modes **0755** and **0644**. Chunk-plan, checkpoint, and debug
|
||||||
|
directories and files use **0700** and **0600**. The operating system's umask
|
||||||
|
may impose stricter output modes. Cache and debug roots may contain sensitive
|
||||||
|
source-derived data, so provision them for one trusted account or service. An
|
||||||
|
output bundle can also contain source content when its JSON output enables
|
||||||
|
evidence publication. Apply an appropriate umask and output-root access policy
|
||||||
|
before enabling that option; the requested output modes alone may not be
|
||||||
|
suitable for transcript-bearing bundles.
|
||||||
|
|
||||||
|
## Run Lifecycle
|
||||||
|
|
||||||
|
Use the [run command](cli.md#run) to start a pipeline. A valid invocation loads
|
||||||
|
and resolves configuration before module preparation and source parsing. It
|
||||||
|
then performs any permitted cache lookup, executes the pipeline, and publishes
|
||||||
|
logical output files only after a successful runner result.
|
||||||
|
|
||||||
|
On success, the command reports the output bundle path. A warning-bearing run
|
||||||
|
still succeeds and reports its warning count on standard error. Errors and
|
||||||
|
their exit classes are defined in the [CLI reference](cli.md#output-streams-and-exit-statuses).
|
||||||
|
|
||||||
|
## Output Bundles
|
||||||
|
|
||||||
|
Each successful run receives a generated safe run identifier and writes beneath:
|
||||||
|
|
||||||
|
~~~
|
||||||
<output-root>/<run-id>/
|
<output-root>/<run-id>/
|
||||||
```
|
~~~
|
||||||
|
|
||||||
The default output root is `./notarius-output`. Use `--output-dir` to choose a
|
The [JSON output contract](integrations/json-output.md) owns the logical files
|
||||||
different root.
|
and their schemas. Before creating the run directory, Notarius validates every
|
||||||
|
logical output path. It refuses an existing run directory without changing it.
|
||||||
|
Files are written atomically; if a later write fails, the newly created partial
|
||||||
|
run directory remains for inspection and is never removed automatically.
|
||||||
|
|
||||||
The `json` output module writes these files:
|
Treat an output bundle as durable user data. Do not use cache-cleanup policy to
|
||||||
|
remove it. An optional accepted chunk map is also durable output and can carry
|
||||||
|
source- or model-derived annotations; its content and compatibility contract
|
||||||
|
are defined in [Accepted Chunk Map](integrations/chunk-map.md). An optional
|
||||||
|
[evidence context](integrations/evidence-context.md) contains source-unit text
|
||||||
|
and metadata. It is not a cache or debug artifact: retain it with the output
|
||||||
|
bundle only for as long as consumers need it, and apply source-content access
|
||||||
|
controls to the entire bundle. Selected lanes may collectively cite most of a
|
||||||
|
transcript, so a broad allowlist can make the evidence artifact nearly as
|
||||||
|
sensitive and large as the source itself.
|
||||||
|
|
||||||
- `index.json`: file index with paths to the manifest, artifact files,
|
## Chunk-Plan Cache
|
||||||
rejected artifacts, and warnings.
|
|
||||||
- `manifest.json`: run manifest with resolved pipeline provenance, module keys,
|
|
||||||
validation status, and timing.
|
|
||||||
- `artifacts/<artifact-type>.json`: approved artifacts grouped by artifact
|
|
||||||
type. For the current D&D spell extractor, this includes
|
|
||||||
`artifacts/dnd.spell_cast.json` when spell-cast artifacts are approved.
|
|
||||||
- `rejected.json`: rejected candidates and validator decisions.
|
|
||||||
- `warnings.json`: warnings reported by pipeline modules or the output encoder.
|
|
||||||
|
|
||||||
Output writes are atomic per file. Logical output file names must be clean,
|
Chunk plans live beneath the selected chunk-plan root:
|
||||||
relative, slash-separated paths and must not contain `..`.
|
|
||||||
|
|
||||||
## Diagnostics Directory
|
~~~
|
||||||
|
<chunk-plan-root>/<source-sha256-hex>/plan.json
|
||||||
|
~~~
|
||||||
|
|
||||||
Diagnostics are written under:
|
One validated canonical plan is active for each source digest. The plan stores
|
||||||
|
boundaries and provenance, not a second copy of the entire source. This
|
||||||
|
source-addressed policy is recorded in [ADR-0005](adr/0005-cache-canonical-chunk-plans-by-source.md).
|
||||||
|
|
||||||
```text
|
The configured cache mode controls one invocation:
|
||||||
<diagnostics-work-dir>/<run-id>/
|
|
||||||
```
|
|
||||||
|
|
||||||
The default diagnostics work directory is `/tmp/notarius`. It can be set with
|
- **auto** looks for a valid active plan. Missing or invalid state causes a new
|
||||||
`diagnostics.work_dir`, `NOTARIUS_WORK_DIR`, or `--diagnostics-dir`.
|
plan to be generated; an accepted new plan is atomically published.
|
||||||
|
- **refresh** skips lookup, generates a plan with the configured chunker, and
|
||||||
|
atomically replaces the active plan after it is accepted.
|
||||||
|
- **bypass** performs no chunk-plan cache I/O. It does not resolve or create a
|
||||||
|
chunk-plan root.
|
||||||
|
|
||||||
Implemented diagnostics artifacts:
|
A reused plan is still materialized and validated against the current source.
|
||||||
|
If a prior plan no longer gives acceptable results, use a refresh run rather
|
||||||
|
than editing cache files. Deleting a plan is recoverable but can repeat costly
|
||||||
|
chunking work.
|
||||||
|
|
||||||
- `invocation.json`: command metadata such as operation, config path, input
|
## Checkpoint Recording, Resume, And Recompute
|
||||||
path, selected lanes, run ID, and pipeline digest when available.
|
|
||||||
- `effective-config.json`: resolved config with API keys redacted.
|
|
||||||
- `resolved-pipeline.json`: resolved module bindings and pipeline digest.
|
|
||||||
- `run-manifest.json`: the same run manifest written to durable output when it
|
|
||||||
is available.
|
|
||||||
- `warnings.json`: warning list.
|
|
||||||
- `run-report.json`: counts, status, output path, diagnostics path, and run ID.
|
|
||||||
- `error.log`: failure message, written after diagnostics directory creation
|
|
||||||
when a run fails.
|
|
||||||
|
|
||||||
`source-document.json` is supported by the diagnostics writer but is not written
|
Checkpoint recording is an explicit configuration choice and is disabled by
|
||||||
by the current CLI run workflow.
|
default. When enabled, each run records stage transitions and the state needed
|
||||||
|
for compatible recovery. A run records checkpoints even when it does not ask
|
||||||
|
to reuse them. Checkpoint payloads can contain source-derived and intermediate
|
||||||
|
application data, so treat the entire root as sensitive.
|
||||||
|
|
||||||
## Retention
|
Checkpoint loading is separate: [**--resume**](cli.md#run) asks a run to reuse
|
||||||
|
compatible recorded work. A resume request fails when checkpoint recording is
|
||||||
|
disabled. Without **--resume**, a recording-enabled run executes normally and
|
||||||
|
does not load checkpoint state. Compatibility includes the resolved pipeline,
|
||||||
|
input, selected lanes, runtime overrides, reference provenance, LLM-profile
|
||||||
|
provenance, and prepared-component fingerprints. A changed identity produces a
|
||||||
|
cold miss; Notarius does not migrate, rewrite, or delete older checkpoint
|
||||||
|
directories.
|
||||||
|
|
||||||
Diagnostics retention is configured with `diagnostics.retention`,
|
Checkpoint state is confined below an identity-specific path:
|
||||||
`NOTARIUS_DIAGNOSTICS_RETENTION`, or the default `auto`.
|
|
||||||
|
|
||||||
- `auto`: keep failed runs and successful runs with warnings; remove successful
|
~~~
|
||||||
warning-free runs.
|
<checkpoint-root>/<pipeline-id>/<input-key>-<source-or-input-digest-prefix>/<pipeline-digest-prefix>/<identity-digest-prefix>/
|
||||||
- `always`: keep every diagnostics run directory.
|
~~~
|
||||||
- `never`: remove successful run directories; failed runs are still retained.
|
|
||||||
|
|
||||||
Unknown retention values are rejected during config validation.
|
### Selective Recompute
|
||||||
|
|
||||||
## Failures
|
[**--recompute-step**](cli.md#run) requires both **--resume** and enabled
|
||||||
|
checkpoint recording. It forces the selected ordered step and every lane that
|
||||||
|
depends on it through generated artifact references. Unrelated lanes remain
|
||||||
|
eligible for reuse.
|
||||||
|
|
||||||
Failures before diagnostics directory creation, such as a missing config file or
|
For an earlier producer required by a forced consumer, Notarius requires a
|
||||||
an unusable diagnostics work directory, are printed to stderr and may not have a
|
compatible accepted normalized artifact. It validates that artifact before
|
||||||
diagnostics run directory.
|
hydrating it and does not silently rerun the producer. If that state is
|
||||||
|
missing, rejected, corrupt, non-canonical, or incompatible, the run stops
|
||||||
|
before its dependent starts. Rerun the required producer deliberately instead
|
||||||
|
of copying or editing checkpoint files.
|
||||||
|
|
||||||
Failures after diagnostics directory creation are printed to stderr and written
|
## Checkpoint Decisions And Recovery
|
||||||
to `error.log`. Depending on where the failure occurred, diagnostics may also
|
|
||||||
include invocation metadata, redacted effective config, resolved pipeline data,
|
|
||||||
the run manifest, warnings, and a run report.
|
|
||||||
|
|
||||||
If durable output writing fails after the pipeline completes, diagnostics are
|
Checkpoint events classify work as **executed**, **reused**,
|
||||||
retained for inspection and may include `run-manifest.json`, `warnings.json`,
|
**forced_recompute**, or **dependency_invalidated**. Their stable reason codes
|
||||||
`run-report.json`, and `error.log`.
|
are written to run diagnostics and provenance. Use the code, not a copied
|
||||||
|
error message, to decide what to repair.
|
||||||
|
|
||||||
## Warnings
|
| Reason code | Recovery meaning |
|
||||||
|
| --- | --- |
|
||||||
|
| **loading_disabled** | This invocation did not permit checkpoint loading. |
|
||||||
|
| **checkpoint_missing**, **checkpoint_path_invalid**, **checkpoint_read_failed**, **checkpoint_decode_failed** | The stored checkpoint could not be located or read safely; normal resume work can execute again. |
|
||||||
|
| **workspace_schema_incompatible**, **identity_mismatch**, **stage_mismatch**, **step_mismatch**, **lane_mismatch**, **module_mismatch** | Stored state belongs to a different compatible scope or identity; allow a fresh run to create new state. |
|
||||||
|
| **status_not_reusable** | The recorded operation did not end in reusable state. |
|
||||||
|
| **dependency_mismatch** | A dependency changed; dependent work is invalidated rather than reused. |
|
||||||
|
| **artifact_payload_invalid**, **artifact_digest_mismatch**, **artifact_codec_incompatible**, **artifact_not_canonical** | A stored artifact cannot safely be hydrated; rerun the producer instead of modifying the cache. |
|
||||||
|
| **checkpoint_reused** | A normal checkpoint passed compatibility checks. |
|
||||||
|
| **accepted_artifact_reused** | A required predecessor's accepted normalized artifact was safely hydrated. |
|
||||||
|
| **recompute_step** | Selective recomputation deliberately forced this work. |
|
||||||
|
|
||||||
A successful run with warnings exits with code `0`, prints a warning count to
|
Reason detail is bounded code-owned text. It is diagnostic information, not a
|
||||||
stderr, and writes warnings to durable output and diagnostics when retained.
|
path-discovery or data-recovery mechanism, and does not contain checkpoint,
|
||||||
|
source, reference, credential, or environment content.
|
||||||
|
|
||||||
The run manifest `validation_status` indicates whether final artifacts were
|
## Debug Bundles
|
||||||
approved or rejected after validation.
|
|
||||||
|
Only a [debug-enabled run](cli.md#run) creates a bundle:
|
||||||
|
|
||||||
|
~~~
|
||||||
|
<debug-root>/<run-id>/
|
||||||
|
summary/
|
||||||
|
trace/
|
||||||
|
~~~
|
||||||
|
|
||||||
|
The summary contains redacted invocation and resolution information plus run,
|
||||||
|
warning, checkpoint, chunk-plan, and terminal reporting artifacts. The trace
|
||||||
|
contains allowlisted application diagnostic records and can include source or
|
||||||
|
derived application data. Neither surface is a cache input. Do not treat a
|
||||||
|
debug bundle as safe to share merely because its configuration summary is
|
||||||
|
redacted.
|
||||||
|
|
||||||
|
Notarius never creates debug state without an explicit request and never
|
||||||
|
automatically deletes a requested bundle. If allocation succeeds, the command
|
||||||
|
reports its path on both success and later failure. A summary, trace, or
|
||||||
|
terminal-report persistence failure fails the command while preserving any
|
||||||
|
already-written diagnostic data for inspection.
|
||||||
|
|
||||||
## Cleanup
|
## Cleanup
|
||||||
|
|
||||||
It is safe to remove specific old run directories after their output and
|
Cleanup is manual and destructive. First inspect the exact leaf directory,
|
||||||
diagnostics are no longer needed:
|
then remove only that leaf; do not use a glob or a parent root as the target.
|
||||||
|
|
||||||
```sh
|
~~~
|
||||||
rm -rf /tmp/notarius/run-1234567890
|
rm -rf -- /srv/notarius/output/run-1721300000000000000-0123456789abcdef0123456789abcdef
|
||||||
rm -rf ./notarius-output/run-1234567890
|
rm -rf -- /srv/notarius/chunk-plans/0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef
|
||||||
```
|
rm -rf -- /srv/notarius/checkpoints/example/seriatim-0123456789abcdef/0123456789abcdef/0123456789abcdef
|
||||||
|
rm -rf -- /srv/notarius/debug/run-1721300000000000000-0123456789abcdef0123456789abcdef
|
||||||
|
~~~
|
||||||
|
|
||||||
Use exact run-directory paths. Avoid broad cleanup commands against parent
|
Deleting output permanently removes user data. Deleting chunk plans or
|
||||||
directories unless they are part of your own operational policy.
|
checkpoints is recoverable but may repeat expensive provider or pipeline work.
|
||||||
|
Deleting a debug bundle removes troubleshooting evidence and a retained copy of
|
||||||
|
application data. Notarius has no cache garbage collector, rollback operation,
|
||||||
|
or automatic cleanup command.
|
||||||
|
|
||||||
## Operational Limits
|
## Operational Limits
|
||||||
|
|
||||||
There is no command to resume a failed run. Re-run `notarius run` after fixing
|
Provider retries and timeouts are supplied by the selected Scriptorium profile.
|
||||||
the cause.
|
Module retry settings and concurrency limits are configuration contracts; see
|
||||||
|
[module bindings](config.md#module-bindings-and-validators) and
|
||||||
Provider retries are limited to the OpenAI-compatible client retry behavior
|
[concurrency](config.md#concurrency-output-cache-and-debug). Extract-worker
|
||||||
configured by the effective LLM profile. There is no separate CLI retry command.
|
limits and actual provider-call limits are independent. Notarius writes local
|
||||||
|
filesystem state only; remote storage, archival, and retention automation are
|
||||||
Notarius writes local files only. Remote storage and archive management are not
|
outside the implemented CLI.
|
||||||
part of the implemented CLI.
|
|
||||||
|
|||||||
@@ -1,210 +1,236 @@
|
|||||||
# Architecture
|
# Architecture
|
||||||
|
|
||||||
This document defines Notarius development policy. It is inward-facing:
|
This document defines the intended high-level architecture of Notarius and the
|
||||||
developers and LLM coding agents should use it to preserve the project's shape,
|
invariants that changes must preserve. Implemented component details belong in
|
||||||
boundaries, and invariants as the code evolves.
|
[Internal Overview](../internal/overview.md) and its linked documents. The
|
||||||
|
reasoning behind significant architectural choices belongs in
|
||||||
|
[ADRs](../adr/).
|
||||||
|
|
||||||
Keep this document concise. It should describe durable architectural rules, not
|
## System Shape
|
||||||
CLI syntax, configuration reference material, module catalogs, or roadmap items.
|
|
||||||
|
|
||||||
## Project Shape
|
Notarius is a small, dependency-light Go application for extracting structured
|
||||||
|
artifacts from source material. It is a general extraction platform whose
|
||||||
|
source formats, extraction domains, validation policies, LLM providers, and
|
||||||
|
output formats are isolated behind explicit boundaries.
|
||||||
|
|
||||||
Notarius is a small, explicit, dependency-light Go application for extracting
|
The application has one fixed pipeline shape:
|
||||||
structured artifacts from source material using modular pipeline stages.
|
|
||||||
|
|
||||||
The application is contract-first but not abstraction-heavy. Add interfaces and
|
|
||||||
extension points when they protect a real boundary:
|
|
||||||
|
|
||||||
- external source formats;
|
|
||||||
- pipeline stage modules;
|
|
||||||
- validators;
|
|
||||||
- LLM providers and runtime plumbing;
|
|
||||||
- output schemas and embedded assets.
|
|
||||||
|
|
||||||
Avoid abstractions that only anticipate hypothetical complexity. Prefer narrow
|
|
||||||
contracts that can be exercised by tests and real modules.
|
|
||||||
|
|
||||||
## Core Invariants
|
|
||||||
|
|
||||||
The framework must remain source-agnostic and domain-agnostic.
|
|
||||||
|
|
||||||
Source-format details belong in input modules. Transcript-specific concepts such
|
|
||||||
as segments, speakers, timestamps, and transcript schemas must not spread into
|
|
||||||
runner, extractor, validator, or LLM framework code.
|
|
||||||
|
|
||||||
Extraction-domain details belong in domain modules. D&D-specific concepts such
|
|
||||||
as spells, NPCs, items, combat turns, and encounters must not spread into core
|
|
||||||
source, runner, or LLM framework packages.
|
|
||||||
|
|
||||||
Extracted facts should be grounded with source references. Source references
|
|
||||||
should point to generic source units, not transcript-only structures. Framework
|
|
||||||
code should preserve source-reference ranges exactly and should not merge or
|
|
||||||
rewrite overlapping ranges unless a module explicitly owns that behavior.
|
|
||||||
|
|
||||||
The application workflow is fixed:
|
|
||||||
|
|
||||||
```text
|
```text
|
||||||
input -> chunk -> extract -> merge -> normalize -> output
|
input -> chunk -> extract -> merge -> normalize -> output
|
||||||
```
|
```
|
||||||
|
|
||||||
These stages should remain explicit in the architecture. Chunking, merging, and
|
Pipelines are configured compositions of this shape. They are not arbitrary
|
||||||
normalization must not be hidden inside domain extractors when they represent
|
DAGs or a general workflow language. Every stage remains explicit; general
|
||||||
general pipeline behavior.
|
chunking, merging, or normalization behavior must not be hidden inside an
|
||||||
|
extractor.
|
||||||
|
|
||||||
Pipelines are fixed-shape templates for this workflow, not arbitrary DAGs or a
|
Input and chunking are pipeline-wide. Each selected artifact lane owns its
|
||||||
general workflow language. Module selection should be configuration- and
|
extract, merge, and normalize stages, and the output stage aggregates the run's
|
||||||
registry-driven, not scattered through conditionals.
|
lane outcomes.
|
||||||
|
|
||||||
## Package Boundaries
|
Notarius is contract-first without being abstraction-heavy. Interfaces and
|
||||||
|
extension points should protect demonstrated boundaries. New abstraction is not
|
||||||
|
itself an architectural goal.
|
||||||
|
|
||||||
Prefer fewer, larger framework packages until a boundary proves itself through
|
## Layers And Dependency Direction
|
||||||
import direction, ownership, test seams, or substantial file size.
|
|
||||||
|
|
||||||
Core packages should contain deterministic models and policy. Framework
|
The application boundary is the composition root and may depend on concrete
|
||||||
packages should contain reusable orchestration and provider plumbing. Concrete
|
implementations. Domain-neutral model and framework layers provide reusable
|
||||||
business logic should live under stage-oriented module packages:
|
policy, contracts, and orchestration. Concrete input, pipeline, output, and
|
||||||
|
validation extensions depend inward on those generic layers.
|
||||||
|
|
||||||
```text
|
Generic layers must not depend on production extensions. Concrete extensions
|
||||||
internal/modules/input/...
|
must not compose the application or take ownership of process behavior. The
|
||||||
internal/modules/chunk/...
|
current packages implementing these layers are inventoried in
|
||||||
internal/modules/extract/...
|
[Internal Overview](../internal/overview.md).
|
||||||
internal/modules/merge/...
|
|
||||||
internal/modules/normalize/...
|
|
||||||
internal/modules/output/...
|
|
||||||
```
|
|
||||||
|
|
||||||
Use short, lowercase, idiomatic Go package names. Avoid package names that repeat
|
The following dependency boundaries are mandatory:
|
||||||
parent-stage context.
|
|
||||||
|
|
||||||
Input modules translate external source formats into the core source model.
|
- extractors and validators do not depend on concrete input adapters;
|
||||||
They may know about external schema details, source-specific metadata, and
|
- provider-specific types do not cross the LLM runtime boundary;
|
||||||
format-specific validation rules. They should not own extraction-domain
|
- external dependency types do not leak across internal package boundaries
|
||||||
decisions.
|
unless that dependency is the package's explicit contract.
|
||||||
|
|
||||||
Extract modules own artifact semantics, prompt usage, structured response schema
|
Shared helpers may support demonstrated common needs, but must not move
|
||||||
selection, validator defaults, and domain-specific interpretation. They should
|
source-format or extraction-domain knowledge into generic framework packages.
|
||||||
depend on framework contracts and core source/artifact types, not concrete input
|
External dependencies require a clear correctness, security, interoperability,
|
||||||
module packages.
|
or complexity benefit.
|
||||||
|
|
||||||
Merge modules combine extracted candidates. Normalize modules reconcile merged
|
## Source And Domain Boundaries
|
||||||
candidates for semantic consistency. Generic behavior may exist for simple
|
|
||||||
artifact types, but domain-specific behavior belongs in modules for the relevant
|
|
||||||
stage.
|
|
||||||
|
|
||||||
Output modules serialize final artifacts and may report warnings out of band.
|
Input modules translate external source formats into the generic source model.
|
||||||
CLI, diagnostics, and reporting layers are responsible for surfacing those
|
Format-specific schemas, fields, and validation remain with the input module
|
||||||
warnings.
|
and its integration contract.
|
||||||
|
|
||||||
|
Framework stages operate on source documents, source units, and source
|
||||||
|
references rather than format-specific structures. A source reference identifies
|
||||||
|
an ordered range of generic source units. Framework code preserves those ranges
|
||||||
|
and does not merge or rewrite them unless a stage module explicitly owns that
|
||||||
|
behavior. Every source unit carries a validated self-reference to its containing
|
||||||
|
document and its own unit ID.
|
||||||
|
|
||||||
|
Extract modules own artifact semantics, prompt use, response schemas, and
|
||||||
|
domain interpretation. Domain-specific concepts remain in the relevant module,
|
||||||
|
validator, shared domain helper, and artifact contract.
|
||||||
|
|
||||||
|
Typed artifact registrations declare one stable artifact kind and exact Go
|
||||||
|
type from extraction through merge, normalization, and semantic validation.
|
||||||
|
Pipeline resolution requires a compatible codec and matching kind-specific
|
||||||
|
variants before a typed lane can be accepted. Framework-owned erasure remains
|
||||||
|
private and must report type incompatibility as an error rather than a panic.
|
||||||
|
|
||||||
|
An artifact kind may additionally provide a typed evidence projection that
|
||||||
|
copies its direct generic source references. Preparation proves that projection
|
||||||
|
matches the artifact codec's exact Go type before retaining it for an output
|
||||||
|
policy. The runner reconstructs evidence only from accepted serialized
|
||||||
|
normalized artifacts, and the output boundary owns any resulting publication.
|
||||||
|
Generic framework code never infers evidence by inspecting domain JSON or
|
||||||
|
depends on domain artifact types.
|
||||||
|
|
||||||
|
Auxiliary references provide context or disambiguation. They are not source
|
||||||
|
evidence and must not be converted into source references.
|
||||||
|
|
||||||
|
## Pipeline Composition And Ownership
|
||||||
|
|
||||||
|
Module selection is configuration- and registry-driven. The framework resolves
|
||||||
|
named pipeline definitions, applies explicit defaults and runtime overrides,
|
||||||
|
and verifies module availability and capabilities before execution. Structural
|
||||||
|
pipeline choices must not be scattered through conditionals or hidden behind
|
||||||
|
ad hoc command flags.
|
||||||
|
|
||||||
|
Resolution validates every selected module and validator option set. A separate
|
||||||
|
preparation boundary then constructs the complete input, chunk, lane,
|
||||||
|
validation, and output implementation set in pipeline order. The runner accepts
|
||||||
|
only that prepared set, so construction and dependency failures occur before
|
||||||
|
source parsing or any other module operation.
|
||||||
|
|
||||||
|
Stage ownership is explicit:
|
||||||
|
|
||||||
|
- input modules convert external material into the generic source model;
|
||||||
|
- chunk modules partition source material for extraction;
|
||||||
|
- extract modules produce domain artifacts from chunks;
|
||||||
|
- merge modules combine accepted extraction outputs;
|
||||||
|
- normalize modules reconcile merged output;
|
||||||
|
- output modules encode accepted results and run outcomes into logical files.
|
||||||
|
|
||||||
|
Chunk modules produce source-addressed chunk plans rather than materialized
|
||||||
|
chunks. The framework validates and materializes those plans into the generic
|
||||||
|
chunk representation before chunk validation and lane execution. Plan reuse is
|
||||||
|
therefore independent of the configured pipeline, module options, references,
|
||||||
|
lanes, validators, and LLM profile: the canonical source digest selects the
|
||||||
|
plan, while the current run still applies its configured chunk validators to
|
||||||
|
the materialized chunks.
|
||||||
|
|
||||||
|
The framework owns orchestration and handoff provenance. Modules return logical
|
||||||
|
results and warnings; they do not own CLI reporting, physical output, cache, or
|
||||||
|
debug roots, durable file placement, or checkpoint and debug lifecycle.
|
||||||
|
|
||||||
|
After pipeline-wide chunking, extraction uses bounded framework concurrency.
|
||||||
|
One run-wide worker pool receives chunk-scoped lane jobs in deterministic
|
||||||
|
chunk-first, lane-second order. A lane may begin its merge and normalize
|
||||||
|
continuation only after all of its extract jobs are terminal; that continuation
|
||||||
|
remains serial within the lane, while bounded continuations for different lanes
|
||||||
|
may overlap. The framework must not create unbounded goroutines per lane or
|
||||||
|
chunk.
|
||||||
|
|
||||||
|
Completion timing does not choose public ordering or errors. The coordinator
|
||||||
|
orders accepted artifacts, warnings, rejections, checkpoint events, and
|
||||||
|
framework errors by stable pipeline scope. Rejections do not cancel unrelated
|
||||||
|
work. A framework error cancels derived work, prevents undispatched work from
|
||||||
|
starting, waits for started work, and prevents output encoding.
|
||||||
|
|
||||||
## Validation
|
## Validation
|
||||||
|
|
||||||
Validators should be independently testable and composable.
|
Validation is a framework-managed boundary around outputs from chunk, extract,
|
||||||
|
merge, and normalize stages. Validators receive immutable stage output
|
||||||
|
and make an explicit whole-output decision: approve, approve with warnings, or
|
||||||
|
reject.
|
||||||
|
|
||||||
Deterministic validators should run before LLM-backed validators when both are
|
Typed artifact validators receive the domain value directly. Chunk validators
|
||||||
present. Validator decision semantics should be explicit: each candidate
|
receive source-zone chunks, while serialized validators receive immutable
|
||||||
artifact evaluated by a validator should receive exactly one decision from that
|
representation bytes and declared schema metadata. A validator registered for
|
||||||
validator.
|
one target or artifact kind cannot satisfy an incompatible selection.
|
||||||
|
|
||||||
LLM-backed review belongs in module-owned validator chains, not in an implicit
|
Rejection is a recorded pipeline outcome, not a framework execution error.
|
||||||
global review phase. Extract and normalize modules may both use deterministic
|
Validator execution failures are framework errors. Rejected output does not
|
||||||
and LLM-backed validators.
|
advance to the next stage.
|
||||||
|
|
||||||
Shared validator runtime mechanics belong in framework code. Concrete validator
|
Default validator chains are production composition policy and are registered
|
||||||
behavior belongs in module or validator implementation packages.
|
centrally by stage and module. Configuration may replace a stage-local default,
|
||||||
|
including with an explicitly empty chain. Configured validator order is
|
||||||
|
authoritative; the framework must not silently reorder it.
|
||||||
|
|
||||||
## LLM Runtime
|
## LLM Boundary
|
||||||
|
|
||||||
LLM provider details belong behind transport-neutral framework contracts.
|
Modules and validators use transport-neutral structured completion contracts.
|
||||||
|
Provider request and response types, authentication, transport behavior, and
|
||||||
|
provider error adaptation remain inside the LLM runtime.
|
||||||
|
|
||||||
Provider-specific HTTP request and response types should stay inside the LLM
|
The caller of the LLM owns prompt selection, prompt inputs, response schema,
|
||||||
runtime package. Prompt construction should stay in extractors, validators, or
|
and interpretation of structured output. Provider adapters do not own source-
|
||||||
shared prompt helpers; provider adapters should not own domain prompt logic.
|
or domain-specific prompt logic.
|
||||||
|
|
||||||
Errors, diagnostics, reports, manifests, and redacted configuration must not
|
LLM calls and other external operations accept cancellation and respect
|
||||||
expose secrets.
|
timeouts. Concurrency control belongs in shared runtime plumbing rather than in
|
||||||
|
individual modules.
|
||||||
|
|
||||||
## Configuration
|
The application-wide LLM scheduler bounds actual provider calls independently
|
||||||
|
of framework worker limits. Every LLM-backed module, retry, and validator uses
|
||||||
|
the single injected scheduled client, including work performed by overlapping
|
||||||
|
lanes.
|
||||||
|
|
||||||
Configuration should make pipeline composition explicit and discoverable.
|
## Configuration And Provenance
|
||||||
|
|
||||||
Centralize configuration loading, precedence, defaults, and validation. Structural
|
Configuration loading, precedence, defaults, environment overrides, redaction,
|
||||||
pipeline choices should come from named pipeline definitions, not ad hoc command
|
and validation are centralized. Named pipeline definitions make structural
|
||||||
flags. Operational overrides may be handled separately when they do not obscure
|
composition explicit and discoverable. Operational overrides are permitted
|
||||||
the configured pipeline structure.
|
when they do not obscure the configured pipeline structure.
|
||||||
|
|
||||||
Module registries should expose module metadata and capabilities without
|
Run preparation fails before stage execution when statically discoverable
|
||||||
requiring module construction. Configuration validation should fail fast when a
|
modules, capabilities, reference bindings, or explicitly selected profiles are
|
||||||
pipeline binds incompatible or unknown modules.
|
invalid or incompatible.
|
||||||
|
|
||||||
Run manifests should record enough resolved pipeline provenance to make a run
|
Run manifests record enough resolved pipeline, module, source, reference, and
|
||||||
auditable after named configuration changes over time.
|
LLM provenance to make a run auditable after configuration changes. Manifests
|
||||||
|
record identities and summaries rather than secret or large payload content.
|
||||||
|
|
||||||
## Dependencies
|
## State, Output, And Safety
|
||||||
|
|
||||||
Prefer the Go standard library where practical.
|
Notarius exposes three filesystem surfaces with independent roots and
|
||||||
|
lifecycle:
|
||||||
|
|
||||||
Use external dependencies only when justified by correctness, security,
|
- output is durable user data; output modules define logical files and the CLI
|
||||||
interoperability, or substantial complexity reduction. Good reasons include
|
owns their placement;
|
||||||
widely used file formats, complex validation behavior, or secure transport
|
- cache is reconstructible state, with separate chunk-plan and checkpoint
|
||||||
handling.
|
families; and
|
||||||
|
- debug is explicitly requested inspection data, combining a redacted summary
|
||||||
|
with a detailed trace.
|
||||||
|
|
||||||
Avoid dependencies for small conveniences. Do not let external dependency types
|
Chunk plans are keyed only by canonical source digest. Configured checkpoint
|
||||||
leak across internal package boundaries unless the dependency is itself the
|
recording is independent of checkpoint reuse; checkpoints are loaded only for
|
||||||
explicit contract of that package.
|
an invocation that explicitly requests resume. Debug is never a cache input and
|
||||||
|
is never created without an explicit request. Pipeline modules receive
|
||||||
|
collaborator interfaces and never physical roots.
|
||||||
|
|
||||||
## State, Files, and Safety
|
Writes are atomic where practical. Paths for writes, moves, overwrites, and
|
||||||
|
deletion must be narrow and explicit. Notarius never automatically deletes
|
||||||
|
output or requested debug bundles; cache cleanup is explicit and recoverable.
|
||||||
|
|
||||||
If the application writes durable state, writes should be atomic where
|
Secrets must not appear in errors, logs, output, cache, debug summaries,
|
||||||
practical. Multi-step workflows should preserve enough diagnostics to support
|
traces, manifests, documentation, examples, or redacted configuration. Debug
|
||||||
inspection after failure.
|
collection is allowlisted to application-owned payloads and must not capture
|
||||||
|
unrelated process environment values or filesystem content. Trace data may
|
||||||
|
contain application data and therefore inherits its sensitivity; operators own
|
||||||
|
access controls and retention. Physical layout and operation are defined in
|
||||||
|
[Operations](../operations.md).
|
||||||
|
|
||||||
Code that deletes, moves, or overwrites files must use narrow, explicit paths.
|
## Architectural Non-Goals
|
||||||
Avoid broad parent-directory operations. Cleanup that can cause data loss must
|
|
||||||
be opt-in.
|
|
||||||
|
|
||||||
## Errors and Logging
|
Notarius does not aim to provide:
|
||||||
|
|
||||||
Errors should be actionable and preserve context. Wrap errors with operation and
|
- an arbitrary workflow graph or general workflow language;
|
||||||
path or resource context. CLI code should convert internal errors into concise
|
- source-format or extraction-domain behavior in generic framework packages;
|
||||||
user-facing messages.
|
- provider-specific contracts exposed to modules;
|
||||||
|
- structural pipeline composition through ad hoc CLI flags;
|
||||||
Errors and logs must not expose secrets. Logs should describe operations,
|
- implicit cross-stage behavior that bypasses the fixed pipeline;
|
||||||
external calls, retries, and failure causes, but should not include large source
|
- abstractions introduced solely for hypothetical future complexity.
|
||||||
or artifact payloads by default.
|
|
||||||
|
|
||||||
Long-running operations should accept `context.Context`. External calls,
|
|
||||||
subprocesses, HTTP requests, storage operations, LLM calls, and multi-stage
|
|
||||||
workflows should respect cancellation and timeouts.
|
|
||||||
|
|
||||||
## Testing
|
|
||||||
|
|
||||||
Core logic should be testable without real external services. Use fakes,
|
|
||||||
fixtures, or local test doubles for input modules, extract modules, validators,
|
|
||||||
and LLM clients where practical.
|
|
||||||
|
|
||||||
Contract-first work should include fake implementations that prove interfaces
|
|
||||||
compose before real modules depend on them.
|
|
||||||
|
|
||||||
Maintain a fixture-driven walking skeleton that exercises the full pipeline with
|
|
||||||
fake modules and fake external clients. This protects stage composition as real
|
|
||||||
modules evolve.
|
|
||||||
|
|
||||||
Important CLI and configuration workflows should have tests. Adapter, extractor,
|
|
||||||
validator, and stage contracts should have focused tests that do not require
|
|
||||||
running the full application unless end-to-end coverage is intentional.
|
|
||||||
|
|
||||||
## Documentation
|
|
||||||
|
|
||||||
Documentation should follow the project documentation policy. Keep user docs
|
|
||||||
focused on implemented behavior. Put future, planned, or aspirational work only
|
|
||||||
under `docs/roadmap/`.
|
|
||||||
|
|
||||||
Core documentation should use generic terms such as source document, source
|
|
||||||
unit, source reference, input adapter, extractor, chunker, merger, normalizer,
|
|
||||||
artifact, validator, and run manifest.
|
|
||||||
|
|
||||||
Source-format details belong in input module or integration docs.
|
|
||||||
Domain-specific extraction details belong in extract module or artifact docs.
|
|
||||||
|
|
||||||
When changing architecture, config, CLI behavior, stage modules, extractor
|
|
||||||
contracts, validator contracts, LLM runtime behavior, or artifact schemas, update
|
|
||||||
the relevant docs and examples in the same change.
|
|
||||||
|
|||||||
@@ -1,139 +0,0 @@
|
|||||||
# Development
|
|
||||||
|
|
||||||
This document defines contributor workflow for Notarius. For architectural
|
|
||||||
invariants and package boundaries, read [Architecture](architecture.md) first.
|
|
||||||
|
|
||||||
## Required Reading
|
|
||||||
|
|
||||||
Before changing the repository, review:
|
|
||||||
|
|
||||||
- [Architecture](architecture.md)
|
|
||||||
- [Documentation Policy](documentation.md)
|
|
||||||
|
|
||||||
Keep current-behavior documentation limited to implemented behavior. Put planned
|
|
||||||
or deferred behavior under `docs/roadmap/`.
|
|
||||||
|
|
||||||
## Repository Layout
|
|
||||||
|
|
||||||
- `cmd/notarius`: executable entry point.
|
|
||||||
- `internal/cli`: CLI parsing, production catalog wiring, config loading, run
|
|
||||||
command orchestration, output writes, and user-facing errors.
|
|
||||||
- `internal/core`: deterministic models and policy for artifacts, source
|
|
||||||
documents, config, and diagnostics.
|
|
||||||
- `internal/framework`: reusable contracts, pipeline orchestration, prompt
|
|
||||||
helpers, validation helpers, and LLM runtime plumbing.
|
|
||||||
- `internal/modules`: concrete input, chunk, extract, merge, normalize, and
|
|
||||||
output modules.
|
|
||||||
- `docs`: policy, user/operator docs, internal docs, integration docs, and
|
|
||||||
roadmap files.
|
|
||||||
- `examples`: maintained, secret-free examples covered by tests where practical.
|
|
||||||
|
|
||||||
## Validation Commands
|
|
||||||
|
|
||||||
Run focused tests for the area changed, then run the broader checks when the
|
|
||||||
change affects shared contracts, CLI behavior, or documentation examples.
|
|
||||||
|
|
||||||
```sh
|
|
||||||
go test ./...
|
|
||||||
go vet ./...
|
|
||||||
go build ./cmd/notarius
|
|
||||||
```
|
|
||||||
|
|
||||||
Useful focused checks:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
go test ./internal/cli
|
|
||||||
go test ./internal/core/config
|
|
||||||
go test ./internal/framework/pipeline
|
|
||||||
go test ./internal/framework/llm
|
|
||||||
go test ./internal/modules/input/seriatim
|
|
||||||
go test ./internal/modules/extract/dnd/spells
|
|
||||||
go test ./internal/modules/output/json
|
|
||||||
```
|
|
||||||
|
|
||||||
## Go Conventions
|
|
||||||
|
|
||||||
- Prefer the standard library unless a dependency is justified by correctness,
|
|
||||||
security, interoperability, or substantial complexity reduction.
|
|
||||||
- Keep package names short, lowercase, and idiomatic.
|
|
||||||
- Preserve import direction: framework and core code must not depend on concrete
|
|
||||||
production modules.
|
|
||||||
- Use `context.Context` for long-running operations and external calls.
|
|
||||||
- Return contextual errors that name the operation and relevant module, path, or
|
|
||||||
resource.
|
|
||||||
- Do not include secrets in errors, logs, diagnostics, manifests, or docs.
|
|
||||||
|
|
||||||
## Adding Config Fields
|
|
||||||
|
|
||||||
Config behavior is centralized under `internal/core/config`.
|
|
||||||
|
|
||||||
When adding a file config field:
|
|
||||||
|
|
||||||
1. Update file config structs and YAML parsing in `file_config.go`.
|
|
||||||
2. Apply the field over defaults in config application code.
|
|
||||||
3. Add validation in `validation.go` when the field has constraints.
|
|
||||||
4. Add environment override support in `env.go` only for operational overrides.
|
|
||||||
5. Update redaction if the field can contain secrets.
|
|
||||||
6. Add focused config tests.
|
|
||||||
7. Update [Configuration](../config.md) and maintained examples when behavior
|
|
||||||
changes.
|
|
||||||
|
|
||||||
Pipeline composition should remain config-driven. Do not add command flags that
|
|
||||||
silently replace structural pipeline definitions.
|
|
||||||
|
|
||||||
## Adding CLI Flags Or Commands
|
|
||||||
|
|
||||||
CLI behavior lives in `internal/cli`.
|
|
||||||
|
|
||||||
When adding CLI surface:
|
|
||||||
|
|
||||||
1. Keep syntax explicit and update usage text.
|
|
||||||
2. Validate arguments before running expensive work.
|
|
||||||
3. Convert internal errors into concise user-facing messages.
|
|
||||||
4. Add CLI tests for success, syntax errors, and failure modes.
|
|
||||||
5. Update [CLI Reference](../cli.md), and update
|
|
||||||
[Operations](../operations.md) or [Troubleshooting](../troubleshooting.md)
|
|
||||||
if run behavior changes.
|
|
||||||
|
|
||||||
## Adding Modules Or Adapters
|
|
||||||
|
|
||||||
Concrete modules live under `internal/modules/<kind>/...` and implement the
|
|
||||||
interfaces in `internal/framework/contracts`.
|
|
||||||
|
|
||||||
For a new production module:
|
|
||||||
|
|
||||||
1. Implement the relevant contract.
|
|
||||||
2. Expose a `ModuleSpec` with the correct module key, module kind, provided
|
|
||||||
capabilities, and required capabilities.
|
|
||||||
3. Expose a `Register` function that registers the module with its registry.
|
|
||||||
4. Add focused module tests for contract behavior, registration, options,
|
|
||||||
validation, and errors.
|
|
||||||
5. Register the module in `internal/cli/catalog.go` only when it is production
|
|
||||||
ready.
|
|
||||||
6. Update internal docs and user-facing docs only for implemented behavior.
|
|
||||||
|
|
||||||
Source-format behavior belongs in input modules and integration docs.
|
|
||||||
Extraction-domain behavior belongs in extract modules and artifact docs.
|
|
||||||
|
|
||||||
## Updating Examples
|
|
||||||
|
|
||||||
Examples must be valid, secret-free, and small.
|
|
||||||
|
|
||||||
- Prefer environment-based secret configuration.
|
|
||||||
- Keep `examples/dnd-spells.config.yml` loadable by CLI tests.
|
|
||||||
- Keep `examples/seriatim-minimal-transcript.json` compatible with the Seriatim
|
|
||||||
adapter.
|
|
||||||
- Do not add expected-output fixtures unless they are validated or have a clear
|
|
||||||
regeneration procedure.
|
|
||||||
|
|
||||||
## Documentation Updates
|
|
||||||
|
|
||||||
Update docs in the same change when behavior changes.
|
|
||||||
|
|
||||||
- CLI syntax: `docs/cli.md`
|
|
||||||
- Config fields and defaults: `docs/config.md`
|
|
||||||
- Output, diagnostics, retention, or recovery: `docs/operations.md`
|
|
||||||
- Common user-facing failures: `docs/troubleshooting.md`
|
|
||||||
- Internal architecture and contracts: `docs/internal/`
|
|
||||||
- External file formats and durable integration contracts: `docs/integrations/`
|
|
||||||
- Future or planned work only: `docs/roadmap/`
|
|
||||||
@@ -1,446 +1,144 @@
|
|||||||
# Go Project Documentation Policy
|
# Documentation Policy
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
|
|
||||||
Project documentation must help five audiences:
|
This policy assigns each documentation topic to one canonical owner. Its goal is
|
||||||
|
to keep Notarius documentation accurate, concise, discoverable, and resistant
|
||||||
1. users who need to run the application;
|
to drift for users, operators, developers, integrators, and LLM coding agents.
|
||||||
2. administrators/operators who need to configure and operate it;
|
|
||||||
3. developers who need to understand and change it safely;
|
|
||||||
4. LLM coding agents that need clear scope, boundaries, and invariants;
|
|
||||||
5. developers and LLM coding agents integrating this project from another codebase.
|
|
||||||
|
|
||||||
Docs should be accurate, concise, task-oriented, and organized by audience. Prefer links to canonical docs over repetition.
|
|
||||||
|
|
||||||
## Core Rules
|
## Core Rules
|
||||||
|
|
||||||
### 1. Keep docs concise
|
### One Canonical Owner
|
||||||
|
|
||||||
Each document should cover a defined scope and only the essentials for that scope.
|
Each authoritative fact belongs in one document. A non-owning document may give
|
||||||
|
a short, stable summary for orientation, but it must link to the canonical owner
|
||||||
|
instead of repeating volatile details.
|
||||||
|
|
||||||
Avoid:
|
Volatile details include commands, flags, configuration fields and defaults,
|
||||||
- long background explanations;
|
module keys, schemas, file names, paths, status codes, retry behavior, and
|
||||||
- repeated reference material;
|
runtime guarantees. If readers could reasonably treat a statement as a
|
||||||
- implementation detail in user-facing docs;
|
contract, maintain it only in the owning document.
|
||||||
- aspirational language outside roadmap docs;
|
|
||||||
- verbose examples where one minimal example is clearer.
|
|
||||||
|
|
||||||
### 2. Document only implemented behavior outside roadmap files
|
### Current And Future Behavior
|
||||||
|
|
||||||
|
Outside `docs/roadmap/`, documentation describes implemented behavior only.
|
||||||
|
Partial features may be described only to their implemented boundary.
|
||||||
|
|
||||||
|
ADRs are the narrow exception: an ADR may record an accepted architectural
|
||||||
|
decision before implementation, but acceptance must not be presented as proof
|
||||||
|
that the behavior exists. The roadmap owns implementation status and sequencing
|
||||||
|
until the decision is implemented. Current architecture, user, operator,
|
||||||
|
integration, and internal documentation are updated when the behavior lands.
|
||||||
|
|
||||||
Unimplemented, planned, aspirational, experimental, or future work may be described only under:
|
### Audience And Detail
|
||||||
|
|
||||||
|
Write for the document's stated audience and include only the detail needed for
|
||||||
|
its owned topic. User and operator docs should not expose implementation detail.
|
||||||
|
Developer docs should link to user-facing and external contracts rather than
|
||||||
|
restate them.
|
||||||
|
|
||||||
- `docs/roadmap/`
|
### Examples
|
||||||
|
|
||||||
No other documentation file, including `README.md`, should describe code, features, modules, stages, commands, config fields, or behaviors that do not currently exist.
|
Complete copyable files belong in `examples/`. Documentation may use the
|
||||||
|
smallest illustrative snippet needed to explain its owned topic, but should link
|
||||||
If a feature is partial, non-roadmap docs may describe only the implemented portion and its current boundary.
|
to maintained examples instead of embedding a second complete copy.
|
||||||
|
|
||||||
### 3. Use canonical homes
|
Examples must be valid, secret-free, and tested where practical. Commands and
|
||||||
|
configuration used in documentation should match the application.
|
||||||
Each type of information should have one canonical location.
|
|
||||||
|
### Security And Privacy
|
||||||
Canonical homes:
|
|
||||||
|
Documentation and examples must not contain real credentials, private keys,
|
||||||
- project purpose and quickstart: `README.md`
|
private environment dumps, sensitive source material, or private infrastructure
|
||||||
- development principles: `docs/policy/architecture.md`
|
details unless intentionally public. Document secret-handling mechanisms, not
|
||||||
- public HTTP API reference: `docs/api.md`
|
secret values.
|
||||||
- configuration reference: `docs/config.md`
|
|
||||||
- CLI reference: `docs/cli.md`
|
## Canonical Ownership
|
||||||
- operations and recovery: `docs/operations.md`
|
|
||||||
- troubleshooting: `docs/troubleshooting.md`
|
| Topic | Canonical owner | Owned content | Content owned elsewhere |
|
||||||
- public API/package consumer guidance: `docs/consumers/`
|
| --- | --- | --- | --- |
|
||||||
- implemented internals: `docs/internal/`
|
| Product orientation and minimal end-to-end quickstart | `README.md` | What Notarius is, why it is useful, one shortest successful invocation, and links onward. | Complete command reference, configuration reference, operational procedures, implementation detail. |
|
||||||
- external protocol, service, and file-format contracts: `docs/integrations/`
|
| Contributor entry point | `docs/development.md` | Task-oriented reading guide, minimal contributor orientation, baseline validation commands, and links to canonical docs. | Package inventory, architecture rules, subsystem behavior, detailed change recipes. |
|
||||||
- future work: `docs/roadmap/`
|
| Current application architecture | `docs/policy/architecture.md` | System shape, normative ownership, dependency direction, architectural boundaries, invariants, safety properties, and non-goals. | Concrete package inventory, implementation mechanics, contributor procedures, decision history, future work. |
|
||||||
- contributor workflow: `docs/policy/development.md`
|
| Documentation organization | `docs/policy/documentation.md` | Documentation ownership, audience boundaries, maintenance rules, and ADR/document lifecycle. | Application architecture or product behavior. |
|
||||||
- copyable examples: `examples/`
|
| Testing policy | `docs/policy/testing.md` | Test philosophy, risk-based sufficiency, test boundaries, doubles, coverage guidance, regression-test policy, and criteria for adding, rewriting, or deleting tests. | Subsystem behavior, application contracts, subsystem-specific test inventories, and implementation plans. |
|
||||||
|
| CLI contract | `docs/cli.md` | Commands, arguments, flags, invocation semantics, and exit codes. | End-to-end operating procedures, configuration field definitions, runtime filesystem layout, module implementation details. |
|
||||||
Other files should summarize briefly and link to the canonical source.
|
| Configuration contract | `docs/config.md` | Discovery and precedence, file schema, fields, defaults, environment overrides, validation rules, and user-selectable module or validator keys. | Complete example files, CLI syntax, runtime state lifecycle, module implementation details. |
|
||||||
|
| Operations | `docs/operations.md` | Runtime workflows, physical filesystem and state layout, output, cache, and debug handling, resume, cleanup, permissions, recovery, and operational limits. | CLI flag syntax, configuration field definitions, logical output schemas, implementation mechanics. |
|
||||||
### 4. Keep examples real
|
| Public HTTP contract, if introduced | `docs/api.md` | Routes, authentication, media types, request and response schemas, status codes, pagination, caching, idempotency, rate limits, and HTTP retry semantics. | Client walkthroughs, upstream or downstream integration internals, implementation detail. |
|
||||||
|
| Consumer guidance, if a public package or API is introduced | `docs/consumers/` | Task-oriented use of the public interface, minimal client examples, and consumer responsibilities. | HTTP wire semantics, external protocol contracts, internal implementation detail. |
|
||||||
Examples should be valid, maintained, and free of secrets.
|
| External and durable integration contracts | `docs/integrations/` | External file formats and protocols, upstream and downstream contracts, logical output bundle paths and schemas, media types, and compatibility behavior. | Physical runtime placement and lifecycle, internal transformations, CLI syntax, configuration defaults. |
|
||||||
|
| Implemented component inventory | `docs/internal/overview.md` | Current packages and components, their implemented responsibilities, and links to focused internal docs. | Normative architecture, contributor reading policy, external contracts. |
|
||||||
Where practical:
|
| Internal component behavior | Other files under `docs/internal/` | Implementation flow, internal collaborators and state transitions, package-local guarantees and failures, and relevant tests. | Global architecture invariants, configuration definitions and defaults, external schemas, operator procedures. |
|
||||||
- example configs should load successfully;
|
| Architectural decision history | `docs/adr/` | Significant decisions, context, alternatives, rationale, consequences, and supersession history. | Current behavior reference, implementation status, task sequencing. |
|
||||||
- example commands should match real CLI syntax;
|
| Future work and implementation status | `docs/roadmap/` | Proposed, accepted, deferred, or rejected work; implementation status; sequencing; and task breakdowns. | Implemented behavior reference and architectural decision rationale. |
|
||||||
- important examples should be covered by tests.
|
| Complete copyable artifacts | `examples/` | Maintained configuration, inputs, and other files intended to be copied or run. | Field-by-field reference, command reference, prose explanation. |
|
||||||
|
|
||||||
## Documentation Profiles
|
Documents that do not exist are required only when the corresponding interface
|
||||||
|
or responsibility exists. Do not create placeholder API, consumer, integration,
|
||||||
All projects require:
|
or operations documents for behavior the application does not have.
|
||||||
|
|
||||||
- `README.md`
|
## Boundary Rules
|
||||||
- `docs/policy/architecture.md`
|
|
||||||
|
### Orientation
|
||||||
Additional docs depend on the project.
|
|
||||||
|
The README owns product orientation. The developer guide routes contributors.
|
||||||
### Small library
|
Architecture owns normative structure. Internal overview owns the current
|
||||||
|
concrete component map. These documents may link to one another but should not
|
||||||
Recommended:
|
maintain parallel package or behavior descriptions.
|
||||||
- `docs/policy/development.md`, if contributor conventions are non-obvious
|
|
||||||
|
### Commands, Configuration, And Operations
|
||||||
### Simple CLI
|
|
||||||
|
CLI documentation answers how to invoke the application. Configuration
|
||||||
Required:
|
documentation answers what settings mean. Operations answers what happens to
|
||||||
- `docs/cli.md`
|
runtime state and how to operate or recover the application. When a workflow
|
||||||
|
crosses these topics, choose the document that owns the task and link to the
|
||||||
Recommended:
|
other contracts.
|
||||||
- `docs/policy/development.md`
|
|
||||||
|
### Contracts And Implementation
|
||||||
### Config-driven CLI
|
|
||||||
|
Integration and API documents define externally observable shapes and
|
||||||
Required:
|
semantics. Internal documents explain how Notarius implements or consumes those
|
||||||
- `docs/cli.md`
|
contracts. Internal docs may name a field, file, or protocol to identify a
|
||||||
- `docs/config.md`
|
dependency, but must link to its canonical contract for the definition.
|
||||||
|
|
||||||
Recommended:
|
### Security Topics
|
||||||
- `examples/`
|
|
||||||
- `docs/policy/development.md`
|
This policy owns what documentation and examples may contain. Architecture owns
|
||||||
|
application security invariants. Configuration owns credential-supply
|
||||||
### Stateful or operator-facing application
|
mechanisms. Operations owns permissions and handling of sensitive runtime
|
||||||
|
artifacts. Internal docs own implementation mechanisms only.
|
||||||
Required:
|
|
||||||
- `docs/cli.md`, if CLI-based
|
## Architecture Decision Records
|
||||||
- `docs/config.md`, if config-driven
|
|
||||||
- `docs/operations.md`
|
Use sequentially numbered ADR filenames such as
|
||||||
|
`0001-record-architecture-decisions.md`. Follow the lightweight Nygard format:
|
||||||
Recommended:
|
|
||||||
- `docs/troubleshooting.md`
|
1. title;
|
||||||
- `examples/`
|
2. status;
|
||||||
- `docs/policy/development.md`
|
3. date;
|
||||||
|
4. context;
|
||||||
### Modular, service-oriented, or orchestration application
|
5. decision;
|
||||||
|
6. alternatives considered;
|
||||||
Required:
|
7. consequences.
|
||||||
- `docs/cli.md`, if CLI-based
|
|
||||||
- `docs/config.md`, if config-driven
|
Treat the decision content of an accepted ADR as immutable. When a decision
|
||||||
- `docs/operations.md`
|
changes, create a new ADR and update the earlier ADR's status to superseded.
|
||||||
- `docs/internal/`
|
Rejected architectural alternatives belong in the ADR; rejected product ideas
|
||||||
- `docs/policy/development.md`
|
belong in the roadmap.
|
||||||
|
|
||||||
Recommended:
|
## Maintenance
|
||||||
- `docs/troubleshooting.md`
|
|
||||||
- validated examples under `examples/`
|
When behavior changes, update its canonical owner in the same change. If
|
||||||
|
ownership moves, remove the old definition and replace it with a link where
|
||||||
### Public HTTP API service
|
navigation remains useful.
|
||||||
|
|
||||||
Required:
|
Before completing documentation work:
|
||||||
- `docs/api.md`
|
|
||||||
- `docs/cli.md`, if CLI-based
|
- verify affected behavior and examples;
|
||||||
- `docs/config.md`, if config-driven
|
- check commands, flags, fields, defaults, schemas, and paths against their
|
||||||
- `docs/operations.md`
|
implementation;
|
||||||
- `docs/internal/`
|
- keep unimplemented behavior in the roadmap, subject to the ADR exception;
|
||||||
- `docs/policy/development.md`
|
- remove stale references and validate links;
|
||||||
|
- confirm that non-owning documents summarize and link rather than redefine;
|
||||||
Recommended:
|
- confirm that no secrets or sensitive private data were added.
|
||||||
- `docs/troubleshooting.md`
|
|
||||||
- `docs/consumers/`, for task-oriented client integration guides
|
|
||||||
- `docs/integrations/`, for upstream/downstream service contracts
|
|
||||||
- validated examples under `examples/`
|
|
||||||
|
|
||||||
### Project with public packages or consumer APIs
|
|
||||||
|
|
||||||
Required:
|
|
||||||
- `docs/consumers/api.md`
|
|
||||||
- one `docs/consumers/pkg-<name>.md` file per public package, if public packages exist
|
|
||||||
|
|
||||||
Recommended:
|
|
||||||
- copyable consumer examples under `examples/`, if practical
|
|
||||||
|
|
||||||
## Required Documents
|
|
||||||
|
|
||||||
### README.md
|
|
||||||
|
|
||||||
**Audience:** users, administrators, operators
|
|
||||||
|
|
||||||
The README is the outward-facing project orientation page.
|
|
||||||
|
|
||||||
It should include, in order:
|
|
||||||
|
|
||||||
1. concise description;
|
|
||||||
2. elevator pitch;
|
|
||||||
3. shortest useful command or usage example;
|
|
||||||
4. links to targeted docs.
|
|
||||||
|
|
||||||
The README should be short. It is not a manual.
|
|
||||||
|
|
||||||
The “shortest useful command” means the simplest command that performs the project’s core use case. (It does not mean `app --help`.)
|
|
||||||
|
|
||||||
### docs/policy/architecture.md
|
|
||||||
|
|
||||||
**Audience:** developers, LLM coding agents
|
|
||||||
|
|
||||||
`docs/policy/architecture.md` is required for every project.
|
|
||||||
|
|
||||||
It is an inward-facing development policy document. It should describe how the project is intended to be built and changed.
|
|
||||||
|
|
||||||
It should include:
|
|
||||||
|
|
||||||
- project shape;
|
|
||||||
- core design principles;
|
|
||||||
- package and boundary philosophy;
|
|
||||||
- state/persistence philosophy, if applicable;
|
|
||||||
- external integration philosophy, if applicable;
|
|
||||||
- error-handling and logging principles;
|
|
||||||
- testing expectations;
|
|
||||||
- documentation expectations;
|
|
||||||
- architectural invariants;
|
|
||||||
- explicit non-goals, if useful.
|
|
||||||
|
|
||||||
Notably, this file should prescribe a core development *policy* that should remain unchanged as the application evolves. It is not a place for details (e.g., CLI flags) that could change over time.
|
|
||||||
|
|
||||||
The contents of `architecture.md` should be trim and concise. LLMs may be directed to review it routinely via AGENTS.md, CLAUDE.md, or similar.
|
|
||||||
|
|
||||||
### docs/api.md
|
|
||||||
|
|
||||||
**Audience:** external HTTP API consumers, developers, LLM coding agents integrating by HTTP
|
|
||||||
|
|
||||||
Required for projects whose primary public interface is HTTP.
|
|
||||||
|
|
||||||
`docs/api.md` is the canonical public HTTP API contract. It should be normative for external consumers and should not be duplicated by README, operations docs, consumer guides, or integration docs.
|
|
||||||
|
|
||||||
It should include:
|
|
||||||
|
|
||||||
1. base URL conventions;
|
|
||||||
2. authentication and authorization behavior, if implemented;
|
|
||||||
3. response envelope;
|
|
||||||
4. supported media types and content negotiation behavior;
|
|
||||||
5. shared query parameters;
|
|
||||||
6. endpoint reference grouped by route family;
|
|
||||||
7. request parameters and validation rules;
|
|
||||||
8. response fields, units, nullability, and optionality;
|
|
||||||
9. error response shape and status codes;
|
|
||||||
10. pagination, caching, rate-limit, idempotency, and retry behavior, if implemented;
|
|
||||||
11. compact request and response examples.
|
|
||||||
|
|
||||||
It must document only implemented endpoints and behavior. Planned endpoints, proposed fields, future filters, and experimental response shapes belong only under `docs/roadmap/`.
|
|
||||||
|
|
||||||
For HTTP API projects, `docs/consumers/` may provide task-oriented client integration guides, but those guides should link to `docs/api.md` for the authoritative endpoint contract.
|
|
||||||
|
|
||||||
### docs/policy/development.md
|
|
||||||
|
|
||||||
**Audience:** developers, LLM coding agents
|
|
||||||
|
|
||||||
Required for projects maintained by humans and LLM coding agents.
|
|
||||||
|
|
||||||
It should include:
|
|
||||||
|
|
||||||
- repository layout;
|
|
||||||
- build/test commands;
|
|
||||||
- coding conventions;
|
|
||||||
- dependency policy;
|
|
||||||
- how to add config fields;
|
|
||||||
- how to add CLI flags;
|
|
||||||
- how to add modules or adapters, if applicable;
|
|
||||||
- how to update examples;
|
|
||||||
- documentation update expectations.
|
|
||||||
|
|
||||||
### docs/config.md
|
|
||||||
|
|
||||||
**Audience:** administrators, operators, advanced users
|
|
||||||
|
|
||||||
Required for applications with configuration files.
|
|
||||||
|
|
||||||
It should include, in order:
|
|
||||||
|
|
||||||
1. config file locations and discovery precedence;
|
|
||||||
2. minimal working config;
|
|
||||||
3. production-oriented config;
|
|
||||||
4. full configuration reference;
|
|
||||||
5. secrets handling, if applicable;
|
|
||||||
6. links to maintained examples.
|
|
||||||
|
|
||||||
The full configuration reference should be canonical.
|
|
||||||
|
|
||||||
### docs/cli.md
|
|
||||||
|
|
||||||
**Audience:** users, administrators, operators
|
|
||||||
|
|
||||||
Required for CLI applications.
|
|
||||||
|
|
||||||
It should include, in order:
|
|
||||||
|
|
||||||
1. shortest useful command;
|
|
||||||
2. command overview;
|
|
||||||
3. complete flag reference;
|
|
||||||
4. common workflows;
|
|
||||||
5. diagnostic or recovery commands, if applicable.
|
|
||||||
|
|
||||||
Explain when commands are useful, not just their syntax.
|
|
||||||
|
|
||||||
### docs/operations.md
|
|
||||||
|
|
||||||
**Audience:** administrators, operators
|
|
||||||
|
|
||||||
Required for applications that maintain state, support resume behavior, run multi-step workflows, write durable artifacts, use remote storage, or require recovery procedures.
|
|
||||||
|
|
||||||
It should cover:
|
|
||||||
|
|
||||||
- normal workflow;
|
|
||||||
- filesystem layout;
|
|
||||||
- remote storage layout, if applicable;
|
|
||||||
- logs and manifests;
|
|
||||||
- resume/retry behavior;
|
|
||||||
- cleanup behavior;
|
|
||||||
- archive/backup behavior;
|
|
||||||
- safe recovery procedures;
|
|
||||||
- operational caveats.
|
|
||||||
|
|
||||||
### docs/troubleshooting.md
|
|
||||||
|
|
||||||
**Audience:** administrators, operators
|
|
||||||
|
|
||||||
Recommended once recurring failure modes exist.
|
|
||||||
|
|
||||||
Each entry should include:
|
|
||||||
|
|
||||||
- symptom;
|
|
||||||
- likely cause;
|
|
||||||
- diagnostic command or inspection step;
|
|
||||||
- safe fix;
|
|
||||||
- relevant links.
|
|
||||||
|
|
||||||
### docs/consumers/
|
|
||||||
|
|
||||||
**Audience:** developers and LLM coding agents integrating this project from another codebase
|
|
||||||
|
|
||||||
Required for projects with public packages, SDKs, client APIs, plugin APIs, or other application-facing integration surfaces.
|
|
||||||
|
|
||||||
This directory describes how an external codebase should consume the project's public API. It should be task-oriented and copyable where useful. It is not the place for internal implementation details or operator procedures.
|
|
||||||
|
|
||||||
For projects whose public API is HTTP, `docs/consumers/` is not required, and it should not duplicate the endpoint reference in `docs/api.md`. If present, it may provide practical integration workflows, client-specific examples, or migration notes that link back to `docs/api.md`.
|
|
||||||
|
|
||||||
`docs/consumers/api.md` should provide the consumer-facing overview and primary implementation workflow. It should include:
|
|
||||||
|
|
||||||
1. intended consumer audience and use cases;
|
|
||||||
2. required inputs supplied by operators or deployment configuration;
|
|
||||||
3. recommended public package or API workflow;
|
|
||||||
4. minimal copyable example;
|
|
||||||
5. consumer responsibilities and boundaries;
|
|
||||||
6. retry, idempotency, or status behavior, if applicable;
|
|
||||||
7. links to package-specific docs and canonical integration contracts.
|
|
||||||
|
|
||||||
Package-specific docs should be named `pkg-<name>.md` and should include:
|
|
||||||
|
|
||||||
1. import path;
|
|
||||||
2. intended use cases;
|
|
||||||
3. primary types and functions needed by consumers;
|
|
||||||
4. minimal examples;
|
|
||||||
5. validation, error, retry, and boundary behavior;
|
|
||||||
6. links to canonical file-format or wire-protocol contracts.
|
|
||||||
|
|
||||||
### docs/internal/
|
|
||||||
|
|
||||||
**Audience:** developers, LLM coding agents
|
|
||||||
|
|
||||||
Required for modular, service-oriented, or orchestration projects.
|
|
||||||
|
|
||||||
This directory describes implemented internal components. It is not the roadmap.
|
|
||||||
|
|
||||||
Use one file per major component where useful.
|
|
||||||
|
|
||||||
Each component doc should include:
|
|
||||||
|
|
||||||
1. purpose;
|
|
||||||
2. inputs and outputs;
|
|
||||||
3. boundaries;
|
|
||||||
4. config fields used;
|
|
||||||
5. external adapters used;
|
|
||||||
6. state or manifest behavior, if applicable;
|
|
||||||
7. skip/resume behavior, if applicable;
|
|
||||||
8. failure behavior;
|
|
||||||
9. tests to inspect before changing;
|
|
||||||
10. architectural invariants.
|
|
||||||
|
|
||||||
### docs/roadmap/
|
|
||||||
|
|
||||||
**Audience:** maintainers, developers, LLM coding agents
|
|
||||||
|
|
||||||
This is the only place for planned, future, aspirational, experimental, or unimplemented work.
|
|
||||||
|
|
||||||
Roadmap docs should clearly distinguish:
|
|
||||||
|
|
||||||
- proposed work;
|
|
||||||
- accepted plans;
|
|
||||||
- deferred ideas;
|
|
||||||
- rejected ideas;
|
|
||||||
- implementation prompts or task breakdowns, if useful.
|
|
||||||
|
|
||||||
Roadmap docs should not be confused with current behavior.
|
|
||||||
|
|
||||||
### docs/integrations/
|
|
||||||
|
|
||||||
**Audience:** developers, LLM coding agents
|
|
||||||
|
|
||||||
Required for projects that depend on external CLIs, APIs, services, protocols, or file formats where the integration contract is important to maintain.
|
|
||||||
|
|
||||||
This directory contains concise, versioned reference notes for external integration contracts. It should document only the parts of the external system that this project actually uses or exposes.
|
|
||||||
|
|
||||||
For public HTTP API services, `docs/integrations/` should document upstream, downstream, storage, protocol, or runtime contracts that the service depends on or bridges. It should not become a second copy of the public HTTP endpoint reference; that belongs in `docs/api.md`.
|
|
||||||
|
|
||||||
Use one file per integration where useful.
|
|
||||||
|
|
||||||
## Examples Directory
|
|
||||||
|
|
||||||
Projects with non-trivial configuration or workflows should include `examples/`.
|
|
||||||
|
|
||||||
Useful examples include:
|
|
||||||
|
|
||||||
- minimal working config;
|
|
||||||
- production-oriented config;
|
|
||||||
- full annotated config;
|
|
||||||
- local development config;
|
|
||||||
- remote/object-storage config;
|
|
||||||
- minimal session/input file.
|
|
||||||
|
|
||||||
Examples should be valid, maintained, tested when practical, and linked from relevant docs.
|
|
||||||
|
|
||||||
## Security and Privacy
|
|
||||||
|
|
||||||
Docs and examples must not include:
|
|
||||||
|
|
||||||
- real API keys;
|
|
||||||
- tokens;
|
|
||||||
- passwords;
|
|
||||||
- private keys;
|
|
||||||
- private environment dumps;
|
|
||||||
- sensitive user data;
|
|
||||||
- raw private transcripts;
|
|
||||||
- private infrastructure details unless intentionally public.
|
|
||||||
|
|
||||||
Document secret-handling mechanisms, not actual secret values.
|
|
||||||
|
|
||||||
## Maintenance Rules
|
|
||||||
|
|
||||||
When docs change, verify the affected behavior.
|
|
||||||
|
|
||||||
Where practical:
|
|
||||||
|
|
||||||
- load example config files in tests;
|
|
||||||
- test CLI examples or command parser behavior;
|
|
||||||
- validate documented flags against real flags;
|
|
||||||
- remove stale references;
|
|
||||||
- update links after renames;
|
|
||||||
- keep roadmap content out of non-roadmap docs.
|
|
||||||
|
|
||||||
If documentation and code disagree, fix the documentation and/or open a roadmap item; do not leave aspirational behavior in current-behavior docs.
|
|
||||||
|
|
||||||
Documentation is complete only when it matches the current code.
|
|
||||||
|
|
||||||
## Documentation Change Checklist
|
|
||||||
|
|
||||||
Before merging documentation changes, verify:
|
|
||||||
|
|
||||||
- README is concise and orientation-focused.
|
|
||||||
- `docs/policy/architecture.md` describes development principles.
|
|
||||||
- `docs/api.md` is the canonical HTTP contract for HTTP API services.
|
|
||||||
- Future work appears only under `docs/roadmap/`.
|
|
||||||
- User-facing docs avoid unnecessary internals.
|
|
||||||
- Consumer-facing docs explain public APIs without duplicating HTTP endpoint or integration contracts.
|
|
||||||
- Developer-facing docs preserve boundaries and invariants.
|
|
||||||
- Config examples match the schema.
|
|
||||||
- CLI examples match real commands and flags.
|
|
||||||
- Defaults appear in the canonical config reference.
|
|
||||||
- No secrets or private data are included.
|
|
||||||
- Links are accurate.
|
|
||||||
|
|||||||
296
docs/policy/testing.md
Normal file
296
docs/policy/testing.md
Normal file
@@ -0,0 +1,296 @@
|
|||||||
|
# Testing Policy
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
|
||||||
|
Our tests exist to make **incorrect changes expensive and correct changes cheap**.
|
||||||
|
|
||||||
|
We do not optimize for test count, line coverage, exhaustive isolation, or the fewest possible tests. We optimize for sufficient confidence in important behavior while imposing as little unnecessary friction as possible on future development.
|
||||||
|
|
||||||
|
## Every test has a cost
|
||||||
|
|
||||||
|
Testing is not an unqualified good. Every test imposes both an immediate cost and a continuing lifetime cost.
|
||||||
|
|
||||||
|
A test must be:
|
||||||
|
|
||||||
|
- written and reviewed;
|
||||||
|
- understood by future maintainers and coding agents;
|
||||||
|
- executed in local and CI workflows;
|
||||||
|
- diagnosed when it fails;
|
||||||
|
- updated when legitimate behavior changes;
|
||||||
|
- maintained as fixtures, APIs, and dependencies evolve; and
|
||||||
|
- removed or rewritten when it becomes redundant, brittle, misleading, or obsolete.
|
||||||
|
|
||||||
|
Tests also create cognitive and architectural friction. They can constrain refactoring, duplicate policy, slow feedback loops, add noise to failures, and cause harmless implementation changes to require unrelated edits across the suite.
|
||||||
|
|
||||||
|
A test is warranted only when the confidence it provides justifies these costs.
|
||||||
|
|
||||||
|
Apply this cost-benefit analysis at two levels:
|
||||||
|
|
||||||
|
1. **Per test:** What realistic defect does this test detect, how consequential would that defect be, and is that protection worth the test's lifetime cost?
|
||||||
|
2. **Across the suite:** Does this collection provide materially more confidence than a smaller, simpler suite would?
|
||||||
|
|
||||||
|
The preferred test suite is a **lean suite that provides sufficient confidence in the risks that matter, without redundant or low-value tests**. We seek sufficient confidence with the least unnecessary testing friction, not the fewest possible tests.
|
||||||
|
|
||||||
|
Some friction is intentional. Tests should make dangerous changes—such as breaking compatibility, corrupting data, violating security boundaries, or reintroducing subtle bugs—require deliberate review. They should not make ordinary internal changes needlessly expensive.
|
||||||
|
|
||||||
|
The cost of a test is not a reason to omit testing by default. Do not cite maintenance cost abstractly. When omitting a plausible test, be able to state why the protected failure is low-risk, already covered, obvious, reversible, or cheaper to detect elsewhere. For consequential, subtle, or difficult-to-observe behavior, the presumption should favor testing.
|
||||||
|
|
||||||
|
## Default testing style
|
||||||
|
|
||||||
|
Use a **classical/Detroit-style** approach:
|
||||||
|
|
||||||
|
- Test observable behavior, resulting state, contracts, and invariants.
|
||||||
|
- Use real internal collaborators when they are fast and deterministic.
|
||||||
|
- Use fakes, stubs, or mocks primarily at expensive, nondeterministic, destructive, or external boundaries.
|
||||||
|
- Prefer package-level behavioral tests over tests coupled to private helpers or internal call sequences.
|
||||||
|
- Treat exact collaborator interactions as testable behavior only when the interaction itself is a requirement.
|
||||||
|
|
||||||
|
Examples of appropriate seams include clocks, randomness, subprocesses, remote APIs, object storage, email, and paid LLM calls.
|
||||||
|
|
||||||
|
## Test execution requirements
|
||||||
|
|
||||||
|
Tests in the default suite must be deterministic, offline, and independent of real credentials. They must not invoke paid APIs or depend on mutable external services. Tests that require live infrastructure must be explicitly opt-in and clearly separated from the default suite.
|
||||||
|
|
||||||
|
Control clocks, randomness, environment variables, and other process-global or machine-specific state when they affect behavior. Tests should be safe to run repeatedly and alongside other tests without depending on execution order or state left by an earlier test.
|
||||||
|
|
||||||
|
## What deserves tests
|
||||||
|
|
||||||
|
Prioritize tests for:
|
||||||
|
|
||||||
|
1. Public and package-level contracts.
|
||||||
|
2. Domain rules and important invariants.
|
||||||
|
3. Boundary conditions and malformed input.
|
||||||
|
4. Failure handling, cancellation, retries, recovery, and partial success.
|
||||||
|
5. Serialization, schemas, compatibility, and round trips.
|
||||||
|
6. Previously observed or plausible regressions.
|
||||||
|
7. Representative integration and end-to-end workflows.
|
||||||
|
|
||||||
|
A package-level contract is behavior relied upon by another package or major collaborator, not every observable detail of a package implementation.
|
||||||
|
|
||||||
|
For behavior involving **data integrity, destructive operations, compatibility, security, concurrency, idempotency, or recovery**, presume that durable tests are required unless the behavior is already credibly protected at another layer.
|
||||||
|
|
||||||
|
Do not add tests merely because a function, branch, or line exists. Do not add a test when the same meaningful risk is already adequately protected elsewhere.
|
||||||
|
|
||||||
|
## Choose the right test boundary
|
||||||
|
|
||||||
|
Test through the narrowest stable boundary that expresses the behavior clearly.
|
||||||
|
|
||||||
|
This is often the package API, but it may instead be:
|
||||||
|
|
||||||
|
- a smaller pure function when dense domain logic is most clearly isolated there;
|
||||||
|
- a package-level operation when several internal collaborators jointly produce the behavior; or
|
||||||
|
- a larger integration boundary when correctness emerges from interaction with a real dependency.
|
||||||
|
|
||||||
|
Do not force all behavior through oversized end-to-end tests. Do not test every private helper merely because it exists. Choose the boundary that gives durable confidence with the least incidental coupling.
|
||||||
|
|
||||||
|
## Test behavior, not implementation
|
||||||
|
|
||||||
|
A test should protect a decision, contract, or invariant—not memorialize the current implementation.
|
||||||
|
|
||||||
|
Before adding or retaining a test, ask:
|
||||||
|
|
||||||
|
> What realistic defect would this test catch?
|
||||||
|
|
||||||
|
A test is suspect when its main purpose is to detect that someone:
|
||||||
|
|
||||||
|
- changed an internal constant;
|
||||||
|
- renamed or split a private helper;
|
||||||
|
- reordered equivalent internal operations;
|
||||||
|
- changed incidental formatting;
|
||||||
|
- replaced one correct algorithm with another; or
|
||||||
|
- refactored internal object structure without changing behavior.
|
||||||
|
|
||||||
|
Refactoring should normally require no test edits unless the refactored structure is itself part of the contract.
|
||||||
|
|
||||||
|
A test can be factually correct and still have negative value. Accurately describing current behavior is not enough; the protected behavior must be important enough to justify the future friction.
|
||||||
|
|
||||||
|
## Expected effects of different changes
|
||||||
|
|
||||||
|
Use the following expectations when evaluating test failures and test maintenance:
|
||||||
|
|
||||||
|
| Change | Expected effect on tests |
|
||||||
|
|---|---|
|
||||||
|
| Internal refactor that preserves behavior | Existing tests should normally remain unchanged and continue to pass. |
|
||||||
|
| Change to an internal default with no contractual significance | Behavioral tests should normally remain unchanged; tests should derive expectations from configuration or relationships rather than duplicate the old value. |
|
||||||
|
| Intentional change to public behavior, policy, schema, or compatibility guarantees | The relevant tests should be reviewed and changed deliberately. |
|
||||||
|
| Accidental violation of a contract or invariant | Tests should fail; fix the production code rather than rewriting the tests to accept the defect. |
|
||||||
|
|
||||||
|
A test failing is not the same as a test needing to be edited. Many tests may correctly fail because of one production defect. The maintenance smell is a correct internal change that requires unrelated expectation updates throughout the suite.
|
||||||
|
|
||||||
|
## Separate mechanism from policy
|
||||||
|
|
||||||
|
Configurable thresholds and defaults must not be duplicated throughout the test suite.
|
||||||
|
|
||||||
|
For example, do not encode an internal concurrency limit indirectly:
|
||||||
|
|
||||||
|
```go
|
||||||
|
// Production policy:
|
||||||
|
const maxConcurrency = 4
|
||||||
|
|
||||||
|
// Brittle test:
|
||||||
|
err := startProcesses(5)
|
||||||
|
require.Error(t, err)
|
||||||
|
```
|
||||||
|
|
||||||
|
Instead, test the mechanism relationally:
|
||||||
|
|
||||||
|
```go
|
||||||
|
const limit = 2
|
||||||
|
runner := NewRunner(limit)
|
||||||
|
|
||||||
|
require.NoError(t, runner.Start(limit))
|
||||||
|
require.ErrorIs(t, runner.Start(limit+1), ErrTooMuchConcurrency)
|
||||||
|
```
|
||||||
|
|
||||||
|
The test should prove:
|
||||||
|
|
||||||
|
- the configured limit is accepted; and
|
||||||
|
- one beyond the configured limit is rejected.
|
||||||
|
|
||||||
|
The production default should be tested exactly only when its literal value is itself a public, operational, safety, protocol, or compatibility requirement.
|
||||||
|
|
||||||
|
Apply the same rule to limits, timeouts, capacities, retry counts, and ranges: test relationships and behavior, not duplicated literals.
|
||||||
|
|
||||||
|
For concurrency limits, test both kinds of behavior when relevant:
|
||||||
|
|
||||||
|
1. **Configuration enforcement:** invalid or excessive requested values are handled correctly.
|
||||||
|
2. **Runtime enforcement:** observed peak concurrency never exceeds the configured limit.
|
||||||
|
|
||||||
|
Use a test-controlled limit and measure the behavior relative to that limit. Do not merely assert today's default value.
|
||||||
|
|
||||||
|
## Avoid semantic duplication across layers
|
||||||
|
|
||||||
|
Each behavior should have a clear test owner.
|
||||||
|
|
||||||
|
- Parser tests own parsing cases.
|
||||||
|
- Validator tests own validation rules.
|
||||||
|
- Domain tests own transformations and invariants.
|
||||||
|
- Adapter tests own external integration behavior.
|
||||||
|
- Orchestrator tests own coordination and failure propagation.
|
||||||
|
- CLI tests own argument and configuration mapping.
|
||||||
|
- End-to-end tests prove that representative assembled workflows work.
|
||||||
|
|
||||||
|
Higher-level tests should not repeat every lower-level case. A single intentional policy change should not require unrelated edits across many test files.
|
||||||
|
|
||||||
|
Tests that are individually reasonable may still be collectively redundant. Evaluate the marginal value of each additional test in light of the protection already provided by the rest of the suite.
|
||||||
|
|
||||||
|
## Use test doubles deliberately
|
||||||
|
|
||||||
|
Choose the least elaborate test double that provides the required control or observation.
|
||||||
|
|
||||||
|
As a default:
|
||||||
|
|
||||||
|
1. Prefer real collaborators when they are fast and deterministic.
|
||||||
|
2. Use small in-memory fakes when realistic stateful behavior is helpful.
|
||||||
|
3. Use stubs when a dependency only needs to provide controlled responses.
|
||||||
|
4. Use mocks when the interaction itself is contractual.
|
||||||
|
|
||||||
|
Mocks are appropriate when the contract includes facts such as:
|
||||||
|
|
||||||
|
- a notification is sent exactly once;
|
||||||
|
- a transaction is committed only after successful writes;
|
||||||
|
- cancellation reaches a subprocess;
|
||||||
|
- an expensive API is called no more than once; or
|
||||||
|
- a security audit event is emitted.
|
||||||
|
|
||||||
|
Do not use mocks merely to isolate every object or reproduce the implementation's call graph.
|
||||||
|
|
||||||
|
## Go-specific guidance
|
||||||
|
|
||||||
|
Use:
|
||||||
|
|
||||||
|
- table-driven tests for meaningful behavioral categories and boundaries;
|
||||||
|
- `t.TempDir()` for real filesystem behavior;
|
||||||
|
- `httptest.Server` for realistic HTTP interactions;
|
||||||
|
- fuzz tests for parsers, normalization, path handling, and broad input spaces;
|
||||||
|
- golden files only when the complete output is intentionally stable;
|
||||||
|
- integration tests where correctness depends on component interaction; and
|
||||||
|
- a small number of representative end-to-end tests.
|
||||||
|
|
||||||
|
Avoid exact error-string assertions unless the wording is itself contractual. Prefer `errors.Is`, `errors.As`, typed errors, or structured error fields.
|
||||||
|
|
||||||
|
At CLI boundaries, prefer exit classifications, structured output, and the smallest stable semantic fragment needed to identify the error. Do not snapshot complete diagnostic wording unless it is contractual.
|
||||||
|
|
||||||
|
Golden-file updates must require an explicit local flag. CI must not update golden files automatically, and reviewers must inspect the semantic diff before accepting an update.
|
||||||
|
|
||||||
|
Keep tests readable and direct. Test helpers and fixture frameworks must earn their own maintenance cost; do not build elaborate test infrastructure for small or isolated needs.
|
||||||
|
|
||||||
|
## Coverage
|
||||||
|
|
||||||
|
Coverage is a diagnostic, not a target.
|
||||||
|
|
||||||
|
Use it to find untested critical branches and unexpectedly weak packages. Do not write low-value tests solely to increase a percentage, and do not infer test quality from coverage alone.
|
||||||
|
|
||||||
|
Pure domain logic will often warrant higher coverage than CLI wiring or external adapters. Uneven coverage is acceptable when it reflects risk.
|
||||||
|
|
||||||
|
Increasing coverage is valuable only when the newly covered behavior protects a meaningful risk at an acceptable cost.
|
||||||
|
|
||||||
|
## Regression tests
|
||||||
|
|
||||||
|
A bug fix should normally include a regression test that fails before the fix and passes afterward.
|
||||||
|
|
||||||
|
Retain the test when the defect could realistically recur and its consequences justify the ongoing cost. Prefer the narrowest durable test of the violated contract or invariant; do not preserve accidental implementation details from the original bug.
|
||||||
|
|
||||||
|
Not every historical bug requires a permanent test. If the underlying design has made recurrence impossible, the test has become redundant, or a stronger invariant test now subsumes it, remove or consolidate it.
|
||||||
|
|
||||||
|
## Deleting or rewriting tests
|
||||||
|
|
||||||
|
Tests are maintained code, not permanent historical artifacts.
|
||||||
|
|
||||||
|
Delete or rewrite a test when its maintenance cost exceeds the confidence it provides.
|
||||||
|
|
||||||
|
Strong candidates include tests that:
|
||||||
|
|
||||||
|
- require updates after harmless internal changes;
|
||||||
|
- directly assert private constants without protecting a real contract;
|
||||||
|
- duplicate the same policy across several layers;
|
||||||
|
- verify mock choreography rather than outcomes;
|
||||||
|
- snapshot large amounts of incidental output;
|
||||||
|
- test trivial private helpers already exercised through stable package behavior;
|
||||||
|
- protect risks already covered more effectively elsewhere;
|
||||||
|
- are flaky, misleading, obsolete, or disproportionately expensive to diagnose; or
|
||||||
|
- no longer correspond to a plausible failure mode.
|
||||||
|
|
||||||
|
Several brittle tests may encode one genuine requirement. Replace them with one durable behavior-level or invariant test rather than preserving all of them.
|
||||||
|
|
||||||
|
Deleting a low-value test can improve the quality of the suite by reducing noise, maintenance burden, and friction around legitimate change.
|
||||||
|
|
||||||
|
## Reviewing a proposed test
|
||||||
|
|
||||||
|
Use the following questions when the value, boundary, or durability of a proposed test is not self-evident. Significant test additions should be reviewable against them, but written answers are not required for every routine test.
|
||||||
|
|
||||||
|
1. What realistic defect would it catch?
|
||||||
|
2. How likely is that defect?
|
||||||
|
3. How consequential would it be?
|
||||||
|
4. Is the behavior already protected elsewhere?
|
||||||
|
5. At which layer should this behavior be owned?
|
||||||
|
6. Does the test assert a durable contract or an incidental implementation detail?
|
||||||
|
7. Could the implementation be refactored without changing the behavior and without editing this test?
|
||||||
|
8. What should cause this test to fail?
|
||||||
|
9. What legitimate changes should not cause this test to fail?
|
||||||
|
10. What ongoing maintenance, execution, and diagnostic cost will the test impose?
|
||||||
|
11. Is there a smaller or more direct test that protects the same risk?
|
||||||
|
|
||||||
|
Do not add the test when its expected lifetime cost exceeds its expected protective value.
|
||||||
|
|
||||||
|
When deciding not to test plausible behavior, record or be able to explain why the risk is low, already protected, obvious, reversible, or cheaper to detect elsewhere.
|
||||||
|
|
||||||
|
## Definition of sufficient
|
||||||
|
|
||||||
|
A test suite is sufficient when:
|
||||||
|
|
||||||
|
- important contracts and invariants are protected;
|
||||||
|
- meaningful boundaries and failure modes are exercised;
|
||||||
|
- realistic and consequential regressions are credibly protected against silent recurrence;
|
||||||
|
- behavior involving data integrity, destructive operations, compatibility, security, concurrency, idempotency, and recovery is credibly protected;
|
||||||
|
- important external boundaries have realistic integration coverage;
|
||||||
|
- representative complete workflows are tested;
|
||||||
|
- failures provide useful signal rather than redundant noise;
|
||||||
|
- legitimate internal changes usually do not require test edits; and
|
||||||
|
- additional tests would mostly repeat existing protection or preserve inconsequential implementation details.
|
||||||
|
|
||||||
|
Sufficiency is a risk judgment, not a coverage percentage or test count. Reassess it as the application, its users, and the consequences of failure evolve.
|
||||||
|
|
||||||
|
The governing rule is:
|
||||||
|
|
||||||
|
> Test heavily where failure is consequential, subtle, or difficult to detect after the fact. Test lightly where failure is obvious, reversible, and inexpensive—and retain no test whose lifetime cost exceeds the confidence it provides.
|
||||||
226
docs/roadmap/evidence.md
Normal file
226
docs/roadmap/evidence.md
Normal file
@@ -0,0 +1,226 @@
|
|||||||
|
# Published Evidence Context
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
Implemented.
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
|
||||||
|
Let downstream consumers build narrative reports from normalized artifacts
|
||||||
|
without separately parsing the original transcript or resolving source-unit
|
||||||
|
references themselves.
|
||||||
|
|
||||||
|
The production JSON output optionally publishes one deterministic, deduplicated
|
||||||
|
evidence-context artifact containing the transcript units relevant to explicitly
|
||||||
|
selected normalized lanes. Existing lane payloads remain the canonical semantic
|
||||||
|
results and retain their precise source references.
|
||||||
|
|
||||||
|
## Desired End State
|
||||||
|
|
||||||
|
When evidence-context publication is enabled, a consumer can:
|
||||||
|
|
||||||
|
1. discover one versioned evidence-context document through `index.json`;
|
||||||
|
2. obtain the union of source units needed to understand evidence cited by the
|
||||||
|
selected normalized lanes;
|
||||||
|
3. distinguish each artifact's direct evidence references from surrounding
|
||||||
|
units included only for narrative context;
|
||||||
|
4. retain speaker, timestamp, and other accepted source-unit metadata needed to
|
||||||
|
interpret the transcript; and
|
||||||
|
5. produce a narrative report without receiving duplicated transcript text in
|
||||||
|
every lane payload.
|
||||||
|
|
||||||
|
This is deterministic output projection. It does not invoke an LLM, change
|
||||||
|
normalization, or make surrounding context part of an artifact's evidence.
|
||||||
|
|
||||||
|
## Configuration Policy
|
||||||
|
|
||||||
|
Evidence publication is configured on the production JSON output module. The
|
||||||
|
intended configuration shape is:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
output:
|
||||||
|
module: json
|
||||||
|
options:
|
||||||
|
evidence_context:
|
||||||
|
enabled: true
|
||||||
|
window_units: 3
|
||||||
|
lanes:
|
||||||
|
- combat-turns
|
||||||
|
- item-events
|
||||||
|
- npc-interactions
|
||||||
|
- npcs
|
||||||
|
- spells
|
||||||
|
```
|
||||||
|
|
||||||
|
- Omitting `evidence_context` disables publication. When the object is present,
|
||||||
|
`enabled` is required.
|
||||||
|
- `enabled: false` accepts no `lanes` or `window_units` fields, preventing
|
||||||
|
silently ignored configuration.
|
||||||
|
- `lanes` is a required, non-empty allowlist of configured final lane IDs when
|
||||||
|
evidence publication is enabled. Values are trimmed, unique, and normalized
|
||||||
|
to lexical order.
|
||||||
|
- `window_units` is a non-negative integer and defaults to `3`. Zero publishes
|
||||||
|
only directly referenced units.
|
||||||
|
- Unknown lanes, duplicate lane IDs, and selected lanes whose artifact kind
|
||||||
|
cannot expose source evidence fail configuration resolution or pipeline
|
||||||
|
preparation.
|
||||||
|
- A selected lane that completes without a normalized output contributes no
|
||||||
|
evidence and does not make an otherwise successful run fail.
|
||||||
|
- Invocation-level lane filtering does not invalidate the configured allowlist.
|
||||||
|
Allowlisted lanes excluded from the effective run contribute nothing, while
|
||||||
|
the evidence document still records the configured allowlist.
|
||||||
|
|
||||||
|
The allowlist is intentional safety and stability policy. Scene descriptions
|
||||||
|
and other broad-range lanes are excluded unless named expressly. Adding a new
|
||||||
|
pipeline lane never silently increases output size or publishes more transcript
|
||||||
|
content.
|
||||||
|
|
||||||
|
## Evidence Collection Boundary
|
||||||
|
|
||||||
|
Evidence collection applies to accepted final normalized artifacts from the
|
||||||
|
selected lanes. It must not inspect arbitrary serialized JSON for fields named
|
||||||
|
`source_ref` or `source_refs`, and the generic JSON output module must not
|
||||||
|
depend on D&D artifact types.
|
||||||
|
|
||||||
|
Artifact-kind registrations expose their source references through an explicit
|
||||||
|
typed projection contract. The framework uses that contract to assemble a
|
||||||
|
domain-neutral evidence request containing:
|
||||||
|
|
||||||
|
- the accepted generic source document;
|
||||||
|
- the selected lane and artifact identities; and
|
||||||
|
- defensive copies of their direct source references.
|
||||||
|
|
||||||
|
The output stage owns publication of the resulting logical artifact. Generic
|
||||||
|
framework code owns range validation, position-based expansion, and union
|
||||||
|
logic. Domain-specific adapters own only the extraction of evidence references
|
||||||
|
from their typed artifacts.
|
||||||
|
|
||||||
|
Both plural-reference artifacts and singular-reference artifacts, such as
|
||||||
|
scene descriptions, can participate through the same projection contract.
|
||||||
|
They do so only when their configured lane is allowlisted.
|
||||||
|
|
||||||
|
## Range Expansion And Deduplication
|
||||||
|
|
||||||
|
For every valid direct source reference:
|
||||||
|
|
||||||
|
1. resolve its endpoints through source-document positions, not numeric
|
||||||
|
unit-ID arithmetic;
|
||||||
|
2. expand the range by `window_units` positions on each side;
|
||||||
|
3. clip the expanded range at document boundaries; and
|
||||||
|
4. union overlapping or contiguous expanded ranges.
|
||||||
|
|
||||||
|
Published contexts and units remain in source-document order. Each source unit
|
||||||
|
appears at most once in a merged context. Original direct references remain
|
||||||
|
unchanged and are associated with their contributing lane IDs so consumers can
|
||||||
|
tell why a context was included.
|
||||||
|
|
||||||
|
The projector must not silently omit or repair an invalid reference that
|
||||||
|
reaches this boundary. Such a value violates the accepted normalized-artifact
|
||||||
|
contract and causes output projection to fail with a content-safe error.
|
||||||
|
|
||||||
|
No implicit coverage limit truncates selected evidence. If the allowlisted
|
||||||
|
lanes collectively cite most or all of a transcript, the evidence document may
|
||||||
|
contain most or all of it. The explicit lane allowlist is the control that
|
||||||
|
prevents a broad lane such as scene descriptions from doing so accidentally.
|
||||||
|
|
||||||
|
## Durable Evidence Artifact
|
||||||
|
|
||||||
|
The JSON bundle gains one optional, non-lane artifact with these durable
|
||||||
|
identities:
|
||||||
|
|
||||||
|
| Property | Value |
|
||||||
|
| --- | --- |
|
||||||
|
| Logical file | `evidence-context.json` |
|
||||||
|
| Index descriptor | `evidence_context` |
|
||||||
|
| Artifact kind | `source/evidence-context` |
|
||||||
|
| Media type | `application/json` |
|
||||||
|
| Schema ID | `notarius.source.evidence_context` |
|
||||||
|
| Schema name | `notarius_source_evidence_context_v1` |
|
||||||
|
| Schema version | `v1` |
|
||||||
|
|
||||||
|
The descriptor in `index.json` carries the artifact and schema identities,
|
||||||
|
analogous to the existing chunk-map descriptor. The artifact is present
|
||||||
|
whenever evidence publication is enabled, including when its context collection
|
||||||
|
is empty.
|
||||||
|
|
||||||
|
The document contains:
|
||||||
|
|
||||||
|
- the source document ID and semantic digest;
|
||||||
|
- the effective window size;
|
||||||
|
- the sorted configured lane allowlist;
|
||||||
|
- an ordered context collection;
|
||||||
|
- each context's expanded start and end unit IDs;
|
||||||
|
- the original direct references and contributing lane IDs covered by that
|
||||||
|
context; and
|
||||||
|
- the ordered accepted source units, including unit ID, kind, text,
|
||||||
|
self-reference, and metadata.
|
||||||
|
|
||||||
|
Expanded context bounds are navigation aids, not citations. The original
|
||||||
|
references embedded in each context remain the authoritative direct evidence.
|
||||||
|
The evidence artifact is discovered separately from lane payloads and does not
|
||||||
|
increase the normalized-lane count reported by the runner or subprocess
|
||||||
|
receipt.
|
||||||
|
|
||||||
|
## Failure And Publication Semantics
|
||||||
|
|
||||||
|
- Evidence projection occurs only after selected normalized outputs are known
|
||||||
|
and before the output encoder returns its logical files.
|
||||||
|
- Projection or encoding failure is an output-stage framework error; the CLI
|
||||||
|
does not publish a partially assembled output bundle.
|
||||||
|
- Rejected or absent lane outputs contribute nothing. Their attempted values
|
||||||
|
and source references must not be published through this artifact.
|
||||||
|
- Context generation is deterministic for the same source document, selected
|
||||||
|
normalized outputs, lane allowlist, and window size.
|
||||||
|
- Existing output, checkpoint, warning, rejection, debug, and subprocess
|
||||||
|
success semantics remain unchanged.
|
||||||
|
|
||||||
|
## Sensitivity And Size
|
||||||
|
|
||||||
|
Unlike the current chunk map, the evidence artifact contains transcript text
|
||||||
|
and source-unit metadata. Enabling it therefore creates additional durable
|
||||||
|
sensitive data and may materially increase bundle size.
|
||||||
|
|
||||||
|
The implemented configuration, operations, integration, and consumer documents
|
||||||
|
state that:
|
||||||
|
|
||||||
|
- evidence publication is opt-in;
|
||||||
|
- output permissions and retention must be appropriate for source content;
|
||||||
|
- selecting broad or numerous lanes can publish most of the transcript; and
|
||||||
|
- the artifact must not contain raw input bytes, LLM prompts or responses,
|
||||||
|
auxiliary reference content, credentials, debug-only data, or filesystem
|
||||||
|
paths.
|
||||||
|
|
||||||
|
## Acceptance Criteria
|
||||||
|
|
||||||
|
- Evidence publication is disabled by default and leaves existing bundles
|
||||||
|
unchanged.
|
||||||
|
- Enabling it requires an explicit non-empty lane allowlist.
|
||||||
|
- References from all selected successful lanes contribute to one deduplicated
|
||||||
|
document.
|
||||||
|
- Non-monotonic unit IDs are expanded and ordered correctly by document
|
||||||
|
position.
|
||||||
|
- Overlapping windows share one ordered copy of each included source unit.
|
||||||
|
- Direct references remain distinguishable from added context.
|
||||||
|
- Scene descriptions cannot contribute unless their lane is explicitly
|
||||||
|
allowlisted.
|
||||||
|
- Invalid selected lanes and unsupported artifact kinds fail before execution;
|
||||||
|
invalid accepted references fail output projection rather than being ignored.
|
||||||
|
- Empty selected-lane results produce a valid empty evidence artifact.
|
||||||
|
- Existing D&D lane schemas, normalized-output counts, and source-reference
|
||||||
|
semantics do not change.
|
||||||
|
- Generic framework and output packages do not depend on D&D types or parse
|
||||||
|
artifact JSON heuristically.
|
||||||
|
- The published contract and operational documentation clearly describe source
|
||||||
|
sensitivity, discovery, compatibility, and retention.
|
||||||
|
|
||||||
|
## Out Of Scope
|
||||||
|
|
||||||
|
- Embedding transcript units directly into each D&D record or lane payload.
|
||||||
|
- Replacing precise source references with expanded context ranges.
|
||||||
|
- Automatically including every configured lane.
|
||||||
|
- An explicit full-transcript publication mode.
|
||||||
|
- LLM summarization, retrieval, ranking, or narrative generation.
|
||||||
|
- Per-record window sizes or lane-specific window sizes.
|
||||||
|
- CLI overrides for evidence configuration.
|
||||||
|
- Reading rejected attempts, debug artifacts, auxiliary references, or prior
|
||||||
|
output bundles as evidence sources.
|
||||||
127
docs/roadmap/future.md
Normal file
127
docs/roadmap/future.md
Normal file
@@ -0,0 +1,127 @@
|
|||||||
|
# Future Work
|
||||||
|
|
||||||
|
Current Notarius behavior is documented in the canonical README, CLI,
|
||||||
|
configuration, operations, internal, and integration docs. This roadmap records
|
||||||
|
future work only. Items are ordered roughly by current value and specificity,
|
||||||
|
not as committed release dates.
|
||||||
|
|
||||||
|
## Near-Term D&D Pipeline
|
||||||
|
|
||||||
|
### Evaluate Spell Extraction And Normalization
|
||||||
|
|
||||||
|
- Evaluate ordinary extraction retries and the completed normalization path
|
||||||
|
against a human-reviewed transcript set before adding repair-aware retries or
|
||||||
|
an LLM-backed semantic validator.
|
||||||
|
- Maintain a small set of human-reviewed transcripts and outputs for prompt,
|
||||||
|
validator, and normalizer development. Treat model-quality review as an
|
||||||
|
iterative human evaluation aid, not a deterministic correctness gate.
|
||||||
|
|
||||||
|
### Evaluate The Shared D&D Scene Plan
|
||||||
|
|
||||||
|
- Reassess whether one shared scene plan provides enough context for NPC,
|
||||||
|
spell, combat, interaction, and scene-description lanes after real-world use.
|
||||||
|
Add more complex chunking only in response to demonstrated failures.
|
||||||
|
|
||||||
|
## Shared Normalization And Quality Work
|
||||||
|
|
||||||
|
### Generic LLM-Assisted Deduplication
|
||||||
|
|
||||||
|
- Add a reusable normalizer that asks an LLM to identify duplicate sets in a
|
||||||
|
list and propose one replacement element for each set.
|
||||||
|
- Define the minimum domain-neutral input contract, initially an ordered list
|
||||||
|
whose elements have stable unique IDs. Artifact-kind registrations or
|
||||||
|
adapters may expose that structure without moving domain rules into the
|
||||||
|
generic package.
|
||||||
|
- Keep mutation deterministic: parse and validate the model's duplicate groups,
|
||||||
|
require every referenced ID to exist, reject overlapping or malformed groups,
|
||||||
|
prevent unrelated insertion or deletion, and apply only approved replacement
|
||||||
|
operations in code.
|
||||||
|
- Preserve provenance needed for audit and downstream validation, and emit
|
||||||
|
warnings describing every collapsed group.
|
||||||
|
- Evaluate batching and context-window limits before applying the normalizer to
|
||||||
|
large artifact collections.
|
||||||
|
|
||||||
|
The model may use its own domain knowledge to judge semantic duplication; the
|
||||||
|
generic implementation is responsible only for the common proposal contract,
|
||||||
|
safety checks, and deterministic application of accepted changes.
|
||||||
|
|
||||||
|
### Validation And Review
|
||||||
|
|
||||||
|
- Add domain validators and production default chains alongside each new D&D
|
||||||
|
artifact.
|
||||||
|
- Add production LLM-backed validators only when a concrete review policy
|
||||||
|
benefits from model judgment and deterministic checks are insufficient.
|
||||||
|
- Add validator diagnostics and timing summaries if operators need more detail
|
||||||
|
than the current [durable output bundle](../integrations/json-output.md)
|
||||||
|
provides.
|
||||||
|
- Add validator compatibility metadata if deployments need config-time proof
|
||||||
|
that a validator is suitable for a particular stage, module, or artifact
|
||||||
|
kind.
|
||||||
|
- Add media-type validators when non-JSON artifact representations are
|
||||||
|
introduced.
|
||||||
|
|
||||||
|
## Further Reference Evolution
|
||||||
|
|
||||||
|
- Make prior-run artifacts easier to bind as references without changing the
|
||||||
|
existing module-facing reference-item contract.
|
||||||
|
- Add structured or parsed references, such as typed NPC registries, rosters,
|
||||||
|
or spell catalogs, when opaque UTF-8 prompt material is no longer sufficient.
|
||||||
|
- Add per-slot or per-chunk inclusion policies so large references are not
|
||||||
|
repeated in every prompt unnecessarily.
|
||||||
|
- Add token budgeting and model context-window management for reference
|
||||||
|
content.
|
||||||
|
- Add reference caching, preprocessing, summarization, embedding, or retrieval
|
||||||
|
only when reference size and observed model behavior justify them.
|
||||||
|
- Extend generated references to prior-run artifacts or derived summaries only
|
||||||
|
after same-run ordered handoffs establish the required provenance and
|
||||||
|
lifecycle semantics.
|
||||||
|
|
||||||
|
## Design Considerations To Revisit
|
||||||
|
|
||||||
|
These concerns are relevant to ordered artifact dependencies but are not
|
||||||
|
committed near-term features.
|
||||||
|
|
||||||
|
### Cross-artifact identity links
|
||||||
|
|
||||||
|
Evaluate whether downstream D&D artifacts should retain canonical NPC IDs from
|
||||||
|
the generated NPC reference in addition to normalized display names. Any such
|
||||||
|
contract must define player-character, unknown-actor, missing-NPC, and
|
||||||
|
superseded-identity behavior before implementation. Deterministic validation
|
||||||
|
may confirm that a linked ID exists in the consumed NPC artifact, but the link
|
||||||
|
must never substitute for transcript evidence that the downstream event
|
||||||
|
occurred.
|
||||||
|
|
||||||
|
### Artifact contract evolution
|
||||||
|
|
||||||
|
Define compatibility and migration policy before generated-reference chains
|
||||||
|
must span multiple schema versions or long-lived historical artifacts. The
|
||||||
|
policy should address stable identifier semantics, which schema changes permit
|
||||||
|
checkpoint reuse, when an older artifact may be decoded or adapted, and when a
|
||||||
|
producer or all dependents must be recomputed. Do not add a general migration
|
||||||
|
framework until an actual contract change requires one.
|
||||||
|
|
||||||
|
## Blue-Sky Platform And Operations
|
||||||
|
|
||||||
|
These ideas are intentionally less specified. Promote one into an earlier
|
||||||
|
section only after a concrete workflow, contract, and priority emerge.
|
||||||
|
|
||||||
|
### Platform Extensions
|
||||||
|
|
||||||
|
- Additional input adapters, such as Markdown or note-export formats.
|
||||||
|
- Additional output encoders.
|
||||||
|
- Concurrent cross-lane entity normalization or broader workflow composition.
|
||||||
|
- Batching or specialized context-window controls for LLM-backed validators.
|
||||||
|
|
||||||
|
### Distribution And Operations
|
||||||
|
|
||||||
|
- Packaged release artifacts for alpha distribution.
|
||||||
|
- A documented versioning and release process.
|
||||||
|
- Optional generated example-output fixtures with a regeneration procedure.
|
||||||
|
- Additional diagnostics or reporting views.
|
||||||
|
|
||||||
|
### Workspace And Storage
|
||||||
|
|
||||||
|
- Default-idempotent run behavior with an explicit force override.
|
||||||
|
- Remote workspace storage.
|
||||||
|
- Workspace garbage collection and archival policies.
|
||||||
|
- Cross-machine checkpoint reuse.
|
||||||
486
docs/roadmap/implementation.md
Normal file
486
docs/roadmap/implementation.md
Normal file
@@ -0,0 +1,486 @@
|
|||||||
|
# Published Evidence Context Implementation Plan
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
Completed.
|
||||||
|
|
||||||
|
## Objective
|
||||||
|
|
||||||
|
Implement the accepted [Published Evidence Context](evidence.md) roadmap as an
|
||||||
|
optional, deterministic extension of the production JSON output. The completed
|
||||||
|
work must publish one deduplicated source-context artifact for an explicit set
|
||||||
|
of successful normalized lanes while leaving existing lane payloads,
|
||||||
|
normalization, checkpointing, subprocess results, and disabled output bundles
|
||||||
|
unchanged.
|
||||||
|
|
||||||
|
Complete the stages below in order. Each stage must leave its affected packages
|
||||||
|
passing before the next begins. Do not implement per-record hydration, implicit
|
||||||
|
all-lane collection, a full-transcript mode, LLM processing, or any other
|
||||||
|
roadmap item marked out of scope.
|
||||||
|
|
||||||
|
## Decisions And Invariants
|
||||||
|
|
||||||
|
- Evidence publication is output policy. Normalizers continue to return
|
||||||
|
semantic artifacts with precise source references and do not receive
|
||||||
|
hydration responsibilities.
|
||||||
|
- The framework operates on the accepted generic `source.SourceDocument` and
|
||||||
|
typed artifact projections. It must not inspect serialized JSON for
|
||||||
|
`source_ref` or `source_refs`, and generic packages must not depend on D&D
|
||||||
|
types.
|
||||||
|
- The selected lane allowlist is explicit, non-empty, and globally addressed by
|
||||||
|
resolved lane ID. Pipeline resolution already guarantees lane IDs are unique
|
||||||
|
across ordered steps.
|
||||||
|
- The framework decodes accepted serialized normalize outputs through their
|
||||||
|
registered artifact codecs before invoking typed evidence projectors. This
|
||||||
|
supports both fresh and checkpoint-reused normalize outputs without retaining
|
||||||
|
a second typed result channel.
|
||||||
|
- Expansion uses source-document positions. Numeric unit IDs are identities,
|
||||||
|
not sequence numbers.
|
||||||
|
- Direct source references are never widened or rewritten. Expanded ranges are
|
||||||
|
context bounds only.
|
||||||
|
- Rejected, failed, and absent normalized lane outputs contribute no evidence.
|
||||||
|
- Evidence output is sensitive durable source content, not cache or debug
|
||||||
|
state. It contains accepted source units only and never raw input bytes,
|
||||||
|
prompts, model responses, auxiliary references, paths, or credentials.
|
||||||
|
- The optional `evidence_context` index field is an additive v1 JSON-bundle
|
||||||
|
change. Existing D&D artifact schemas and the subprocess receipt do not
|
||||||
|
change.
|
||||||
|
|
||||||
|
## Stage 1: Add Typed Evidence Capability And Resolve Output Policy
|
||||||
|
|
||||||
|
Add a dedicated artifact-evidence registry under the pipeline framework:
|
||||||
|
|
||||||
|
- `pipeline.ArtifactEvidenceRegistry` stores one typed projector per artifact
|
||||||
|
kind.
|
||||||
|
- `pipeline.ArtifactEvidenceProjector[T]` is
|
||||||
|
`func(T) []source.SourceRef`.
|
||||||
|
- `pipeline.RegisterArtifactEvidence[T](registry, kind, projector)` accepts a
|
||||||
|
non-empty kind and non-nil projector, records the exact Go type for `T`, and
|
||||||
|
rejects duplicate kinds.
|
||||||
|
- The erased projection boundary checks the exact registered type, invokes the
|
||||||
|
projector, and returns a defensive copy of its references.
|
||||||
|
- The registry exposes only the discovery and projection operations required by
|
||||||
|
resolution, preparation, and execution; do not expose its mutable entries.
|
||||||
|
|
||||||
|
Add the registry to `pipeline.Registries` and `pipeline.ModuleCatalog`, including
|
||||||
|
CLI catalog conversion, production construction, empty-set detection, and
|
||||||
|
test registry helpers. A nil evidence registry remains valid when evidence
|
||||||
|
publication is disabled. Production construction and the D&D registrar require
|
||||||
|
and populate it.
|
||||||
|
|
||||||
|
Define these framework-level output-policy contracts:
|
||||||
|
|
||||||
|
```go
|
||||||
|
type EvidenceContextPolicy struct {
|
||||||
|
Enabled bool
|
||||||
|
WindowUnits int
|
||||||
|
LaneIDs []string
|
||||||
|
}
|
||||||
|
|
||||||
|
type EvidenceContextPolicyProvider interface {
|
||||||
|
EvidenceContextPolicy() EvidenceContextPolicy
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Provider and prepared-pipeline boundaries defensively copy `LaneIDs`.
|
||||||
|
|
||||||
|
Add an optional output-profile option-validation callback to
|
||||||
|
`OutputEncoderRegistry`:
|
||||||
|
|
||||||
|
```go
|
||||||
|
type OutputProfileOptionContext struct {
|
||||||
|
LaneIDs []string
|
||||||
|
}
|
||||||
|
|
||||||
|
type OutputProfileOptionValidator func(
|
||||||
|
OutputProfileOptionContext,
|
||||||
|
map[string]any,
|
||||||
|
) error
|
||||||
|
```
|
||||||
|
|
||||||
|
Add `RegisterBuilderWithProfileValidation(spec, validateOptions,
|
||||||
|
validateProfile, builder)` and make existing output registration methods
|
||||||
|
delegate to it with no profile callback. The registry passes defensive copies
|
||||||
|
to validation. The callback receives the complete configured lane-ID set before
|
||||||
|
invocation-level `--only` filtering. The production JSON output uses it only to
|
||||||
|
prove that every configured evidence lane exists. Keep the extension generic:
|
||||||
|
the pipeline supplies lane identities, while the output module interprets its
|
||||||
|
own options. Build that lane set from the normalized legacy-or-steps profile
|
||||||
|
before selection and reject duplicate configured lane IDs through the existing
|
||||||
|
pipeline identity rules.
|
||||||
|
|
||||||
|
Extend the production JSON output options with the nested
|
||||||
|
`evidence_context` object:
|
||||||
|
|
||||||
|
- omission disables the feature;
|
||||||
|
- `enabled` is required when the object is present;
|
||||||
|
- `enabled: false` permits no `lanes` or `window_units` fields;
|
||||||
|
- `enabled: true` requires a non-empty `lanes` array;
|
||||||
|
- lane values are strings, trimmed, non-empty, unique after trimming, and
|
||||||
|
normalized to lexical order;
|
||||||
|
- `window_units` is an optional non-negative integer with default `3`; and
|
||||||
|
- outer and nested unknown fields and incompatible YAML value types remain
|
||||||
|
strict configuration errors.
|
||||||
|
|
||||||
|
The JSON encoder implements the policy provider from its decoded immutable
|
||||||
|
options. Pipeline resolution invokes its profile validator against all
|
||||||
|
configured steps, so an unknown evidence lane fails even when another lane is
|
||||||
|
selected with `--only`.
|
||||||
|
|
||||||
|
During `pipeline.Prepare`, after constructing the output encoder:
|
||||||
|
|
||||||
|
1. obtain and defensively normalize an enabled policy;
|
||||||
|
2. intersect its configured IDs with the effective prepared lanes, treating
|
||||||
|
allowlisted lanes removed by invocation-level filtering as inactive;
|
||||||
|
3. require an artifact-evidence registration for each active lane kind;
|
||||||
|
4. prove that its projector Go type exactly matches the active lane's registered
|
||||||
|
artifact codec type; and
|
||||||
|
5. retain an immutable private evidence plan on `PreparedPipeline`.
|
||||||
|
|
||||||
|
Duplicate or empty provider values, a missing evidence registry for an active
|
||||||
|
lane, unsupported active artifact kinds, and type mismatches fail preparation
|
||||||
|
with pipeline/output/lane context. A disabled or non-participating output
|
||||||
|
encoder creates no evidence plan and preserves existing preparation behavior.
|
||||||
|
The private plan retains both the full configured allowlist for publication and
|
||||||
|
the active lane/projector intersection for execution.
|
||||||
|
|
||||||
|
Register D&D evidence projectors for all six current artifact kinds. Each
|
||||||
|
projector returns copies of the artifact's direct references in record order:
|
||||||
|
spells, NPCs, combat turns, item events, NPC interactions, and the singular
|
||||||
|
reference from each scene description. Scene descriptions gain capability but
|
||||||
|
remain excluded unless their configured lane ID is allowlisted.
|
||||||
|
|
||||||
|
Stage tests:
|
||||||
|
|
||||||
|
- Registry tests cover nil, blank, duplicate, exact-type, defensive-copy, and
|
||||||
|
deterministic discovery behavior.
|
||||||
|
- JSON option tests cover disabled, enabled/default-window, explicit zero
|
||||||
|
window, normalization, duplicates, unknown fields, and invalid types.
|
||||||
|
- Preparation tests cover selected lanes across steps, unknown lanes,
|
||||||
|
unsupported kinds, projector/codec type mismatch, disabled behavior, and
|
||||||
|
defensive policy ownership.
|
||||||
|
- Resolution/preparation tests prove a valid allowlist survives `--only`, an
|
||||||
|
excluded lane contributes no active projector, and a genuinely unknown
|
||||||
|
configured lane still fails profile resolution.
|
||||||
|
- D&D registration tests prove every production D&D artifact kind has the
|
||||||
|
expected evidence capability without testing individual field loops
|
||||||
|
redundantly.
|
||||||
|
- One table-driven D&D projector test supplies representative values for all
|
||||||
|
six artifact kinds and proves plural and singular references are copied
|
||||||
|
without aliasing or semantic rewriting.
|
||||||
|
|
||||||
|
Stage completion:
|
||||||
|
|
||||||
|
- `go test ./internal/framework/pipeline`
|
||||||
|
- `go test ./internal/modules/generic/output/json`
|
||||||
|
- `go test ./internal/modules/dnd/register`
|
||||||
|
- `go test ./internal/cli`
|
||||||
|
|
||||||
|
## Stage 2: Define And Build The Evidence-Context Artifact
|
||||||
|
|
||||||
|
Add a domain-neutral `internal/framework/evidencecontext` package that owns the
|
||||||
|
durable model, JSON Schema, strict codec, projection algorithm, and these exact
|
||||||
|
identities:
|
||||||
|
|
||||||
|
- artifact kind `source/evidence-context`;
|
||||||
|
- media type `application/json`;
|
||||||
|
- schema ID `notarius.source.evidence_context`;
|
||||||
|
- schema name `notarius_source_evidence_context_v1`; and
|
||||||
|
- schema version `v1`.
|
||||||
|
|
||||||
|
Use these package-level model and build contracts:
|
||||||
|
|
||||||
|
```go
|
||||||
|
type Document struct {
|
||||||
|
SourceID string
|
||||||
|
SourceDigest string
|
||||||
|
WindowUnits int
|
||||||
|
SelectedLanes []string
|
||||||
|
Contexts []Context
|
||||||
|
}
|
||||||
|
|
||||||
|
type Context struct {
|
||||||
|
ContextRef source.SourceRef
|
||||||
|
EvidenceRefs []EvidenceRef
|
||||||
|
Units []source.SourceUnit
|
||||||
|
}
|
||||||
|
|
||||||
|
type EvidenceRef struct {
|
||||||
|
LaneID string
|
||||||
|
SourceRef source.SourceRef
|
||||||
|
}
|
||||||
|
|
||||||
|
type LaneEvidence struct {
|
||||||
|
LaneID string
|
||||||
|
SourceRefs []source.SourceRef
|
||||||
|
}
|
||||||
|
|
||||||
|
type BuildRequest struct {
|
||||||
|
Source *source.SourceDocument
|
||||||
|
WindowUnits int
|
||||||
|
SelectedLanes []string
|
||||||
|
LaneEvidence []LaneEvidence
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Apply the JSON field names shown below. Provide `Build(BuildRequest)`,
|
||||||
|
`Serialize(BuildRequest)`, and a `Codec` with the same identity/encode/decode
|
||||||
|
responsibilities as the chunk-map codec.
|
||||||
|
|
||||||
|
The v1 payload has this exact shape:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"source_id": "session-alpha",
|
||||||
|
"source_digest": "sha256:0000000000000000000000000000000000000000000000000000000000000000",
|
||||||
|
"window_units": 0,
|
||||||
|
"selected_lanes": ["npcs"],
|
||||||
|
"contexts": [
|
||||||
|
{
|
||||||
|
"context_ref": {
|
||||||
|
"source_id": "session-alpha",
|
||||||
|
"start_unit_id": 13,
|
||||||
|
"end_unit_id": 13
|
||||||
|
},
|
||||||
|
"evidence_refs": [
|
||||||
|
{
|
||||||
|
"lane_id": "npcs",
|
||||||
|
"source_ref": {
|
||||||
|
"source_id": "session-alpha",
|
||||||
|
"start_unit_id": 13,
|
||||||
|
"end_unit_id": 13
|
||||||
|
}
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"units": [
|
||||||
|
{
|
||||||
|
"id": 13,
|
||||||
|
"kind": "transcript_segment",
|
||||||
|
"text": "The party meets Rowan.",
|
||||||
|
"ref": {
|
||||||
|
"source_id": "session-alpha",
|
||||||
|
"start_unit_id": 13,
|
||||||
|
"end_unit_id": 13
|
||||||
|
}
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
All displayed fields are required. `selected_lanes`, `contexts`,
|
||||||
|
`evidence_refs`, and `units` encode as arrays rather than `null`; `contexts`
|
||||||
|
may be empty. Each unit uses the existing `source.SourceUnit` JSON shape with
|
||||||
|
required `id`, `kind`, `text`, and `ref`, plus optional JSON-shaped `metadata`.
|
||||||
|
Fixed objects reject unknown fields; metadata remains an arbitrary JSON object.
|
||||||
|
|
||||||
|
The builder accepts the validated source document, selected lane IDs, effective
|
||||||
|
window, and lane-attributed direct references, then:
|
||||||
|
|
||||||
|
1. requires a non-negative window and a non-empty, trimmed, unique selected
|
||||||
|
lane set, then stores that set in lexical order;
|
||||||
|
2. requires every `LaneEvidence.LaneID` to belong to the selected set;
|
||||||
|
3. validates the source document, recomputes its semantic digest, and requires
|
||||||
|
it to equal `SourceDocument.Digest`;
|
||||||
|
4. validates every direct reference against one `source.DocumentIndex`;
|
||||||
|
5. deduplicates exact `(lane_id, source_ref)` contributions;
|
||||||
|
6. resolves endpoints to document positions;
|
||||||
|
7. expands each side without integer overflow and clips at document bounds;
|
||||||
|
8. sorts by expanded document position with deterministic lane/reference
|
||||||
|
tie-breakers;
|
||||||
|
9. merges overlapping or position-contiguous expanded intervals;
|
||||||
|
10. unions and deterministically sorts each merged context's direct
|
||||||
|
contributions; and
|
||||||
|
11. deep-clones the corresponding source units and JSON-shaped metadata.
|
||||||
|
|
||||||
|
Contexts are disjoint and ordered by document position, so a source unit occurs
|
||||||
|
at most once in the document. `context_ref` identifies the first and last
|
||||||
|
included units; `evidence_refs` retains only original citations. An empty
|
||||||
|
reference collection produces the same source identity, window, sorted
|
||||||
|
allowlist, and an explicit empty contexts array.
|
||||||
|
|
||||||
|
Projection failures identify only structural scope such as lane and reference
|
||||||
|
position. They must not include source text, metadata values, raw serialized
|
||||||
|
artifacts, or unrelated paths.
|
||||||
|
|
||||||
|
The codec must validate its model before encoding, produce deterministic JSON,
|
||||||
|
strictly decode the checked-in schema, and return independently owned values.
|
||||||
|
The JSON output encoder remains responsible for pretty-printing the logical
|
||||||
|
file with its standard trailing newline. Follow the existing chunk-map
|
||||||
|
package's separation between model, builder, codec, schema asset, and contract
|
||||||
|
tests where useful, without coupling the two artifact formats.
|
||||||
|
|
||||||
|
Stage tests:
|
||||||
|
|
||||||
|
- A table-driven builder suite covers zero and nonzero windows, boundary
|
||||||
|
clipping, non-monotonic unit IDs, separate gaps, overlapping and contiguous
|
||||||
|
windows, duplicate contributions, multiple lanes, stable ordering, empty
|
||||||
|
contexts, invalid selected/contributing lanes, source-digest mismatch, and
|
||||||
|
invalid references.
|
||||||
|
- Ownership tests prove output mutation cannot affect the source document or
|
||||||
|
projector inputs, including nested metadata.
|
||||||
|
- Codec tests cover round trip, required arrays, schema identity, malformed and
|
||||||
|
trailing JSON, unknown fixed fields, invalid ordering/ranges, mismatched
|
||||||
|
source identities, and independently owned decoded metadata.
|
||||||
|
- Use structured assertions and a compact valid fixture; do not add a large
|
||||||
|
transcript golden file.
|
||||||
|
|
||||||
|
Stage completion:
|
||||||
|
|
||||||
|
- `go test ./internal/framework/evidencecontext`
|
||||||
|
|
||||||
|
## Stage 3: Integrate Projection With Runner Output
|
||||||
|
|
||||||
|
Extend `contracts.OutputRequest` with an optional
|
||||||
|
`EvidenceContext *SerializedArtifact` field and clone it at every ownership
|
||||||
|
handoff, following the existing chunk-map pointer pattern.
|
||||||
|
|
||||||
|
After lane execution and final manifest population, but before invoking the
|
||||||
|
output encoder, the runner must:
|
||||||
|
|
||||||
|
1. skip all work when the prepared evidence plan is absent;
|
||||||
|
2. index accepted `NormalizeOutputs` by their globally unique lane IDs and fail
|
||||||
|
on an internal duplicate rather than silently overwrite it;
|
||||||
|
3. for each selected lane with an output, verify its source and artifact kind,
|
||||||
|
decode it through the prepared artifact codec registry, and invoke the
|
||||||
|
prepared typed projector;
|
||||||
|
4. build and serialize the evidence document through the evidence-context
|
||||||
|
package; and
|
||||||
|
5. pass a defensive serialized-artifact copy to the output encoder.
|
||||||
|
|
||||||
|
Selected lanes without normalized output contribute nothing. Normalize
|
||||||
|
rejections remain successful pipeline outcomes; evidence projection does not
|
||||||
|
inspect rejected candidates. An invalid accepted reference, incompatible
|
||||||
|
serialized artifact, projection type failure, or evidence serialization failure
|
||||||
|
is an output-stage framework error before logical files are returned or
|
||||||
|
physically published.
|
||||||
|
|
||||||
|
At this external-content consumption boundary, do not propagate artifact-codec
|
||||||
|
or metadata-cloning errors with `%w` when their text could contain artifact
|
||||||
|
fields or source metadata. Return fixed, actionable categories scoped by lane
|
||||||
|
and operation; detailed codec errors remain available to direct trusted
|
||||||
|
callers and their focused tests.
|
||||||
|
|
||||||
|
Add an allowlisted debug summary containing only evidence artifact identity,
|
||||||
|
selected lanes, window, context count, unit count, and source digest. Do not
|
||||||
|
duplicate transcript text or source-unit metadata into a new evidence-specific
|
||||||
|
debug envelope. Existing normalized-output debug behavior remains unchanged.
|
||||||
|
|
||||||
|
The output artifact is not a normalized lane, generated reference, checkpoint,
|
||||||
|
or manifest normalized-output entry. It does not alter normalized-output,
|
||||||
|
rejection, or warning counts. Resume continues to reuse normalized checkpoints;
|
||||||
|
evidence is deterministically rebuilt during the always-executed output stage.
|
||||||
|
|
||||||
|
Stage tests:
|
||||||
|
|
||||||
|
- Runner tests use a real codec, evidence projector, and small capturing output
|
||||||
|
encoder to prove selected-lane union, absent/rejected lane omission, invalid
|
||||||
|
accepted-reference failure, output-request defensive ownership, and no work
|
||||||
|
when disabled.
|
||||||
|
- Include one checkpoint-reused normalized-output case to prove evidence is
|
||||||
|
reconstructed identically without retaining typed normalize values.
|
||||||
|
- Confirm projection failures prevent output encoding and return a failed
|
||||||
|
manifest without changing rejection semantics.
|
||||||
|
|
||||||
|
Stage completion:
|
||||||
|
|
||||||
|
- `go test ./internal/framework/pipeline`
|
||||||
|
|
||||||
|
## Stage 4: Publish Through The JSON Bundle
|
||||||
|
|
||||||
|
Teach the production JSON encoder to recognize the optional evidence-context
|
||||||
|
artifact, verify its exact kind, media type, schema identity, schema digest, and
|
||||||
|
payload validity through the evidence-context codec, and emit:
|
||||||
|
|
||||||
|
- logical file `evidence-context.json`; and
|
||||||
|
- optional `index.json` descriptor field `evidence_context`.
|
||||||
|
|
||||||
|
The descriptor uses the same six fields as `chunk_map`:
|
||||||
|
`artifact_kind`, `file`, `media_type`, `schema_id`, `schema_name`, and
|
||||||
|
`schema_version`. Refactor the encoder's private descriptor representation only
|
||||||
|
as needed to share that shape; do not change the existing `chunk_map` wire
|
||||||
|
contract. Evidence output is ordered with the encoder's other fixed logical
|
||||||
|
files, remains a non-lane artifact, and is present with an empty contexts array
|
||||||
|
when enabled but no selected lane produces references.
|
||||||
|
|
||||||
|
The JSON encoder's validation boundary returns a fixed content-safe evidence
|
||||||
|
artifact error rather than propagating decoder or schema diagnostics that could
|
||||||
|
echo transcript text or metadata. Direct evidence-context codec tests retain
|
||||||
|
detailed structural errors.
|
||||||
|
|
||||||
|
Update the maintained complete D&D configuration to enable evidence context
|
||||||
|
with window `3` for `item-events`, `npcs`, `spells`, `combat-turns`, and
|
||||||
|
`npc-interactions`. Deliberately omit `scene-descriptions`. Keep the minimal
|
||||||
|
configuration disabled by omission.
|
||||||
|
|
||||||
|
Stage tests:
|
||||||
|
|
||||||
|
- JSON encoder tests own descriptor shape, exact logical filename, identity
|
||||||
|
checking, empty evidence publication, and disabled bundle stability.
|
||||||
|
- One assembled production D&D test uses multiple selected lanes with
|
||||||
|
overlapping references and non-monotonic unit IDs, decodes the published
|
||||||
|
artifact through its production codec, and proves union/deduplication and
|
||||||
|
scene-description exclusion.
|
||||||
|
- A second narrow case explicitly allowlists a scene-description lane to prove
|
||||||
|
capability is opt-in rather than hard-coded exclusion.
|
||||||
|
- Existing index, chunk-map, lane, manifest, warning, and rejection tests remain
|
||||||
|
the owners of their current formats; do not repeat their full matrices.
|
||||||
|
|
||||||
|
Stage completion:
|
||||||
|
|
||||||
|
- `go test ./internal/modules/generic/output/json`
|
||||||
|
- `go test ./internal/modules/dnd/...`
|
||||||
|
- `go test ./internal/modules/integration`
|
||||||
|
- `go test ./internal/cli`
|
||||||
|
|
||||||
|
## Stage 5: Publish Current-Behavior Documentation
|
||||||
|
|
||||||
|
After implementation and behavioral tests pass, update canonical documentation:
|
||||||
|
|
||||||
|
- `docs/config.md` owns the nested JSON output options, defaults, strict
|
||||||
|
validation, required lane allowlist, and a small configuration snippet.
|
||||||
|
- A new `docs/integrations/evidence-context.md` owns the complete v1 payload,
|
||||||
|
identities, direct-evidence versus context semantics, ordering,
|
||||||
|
compatibility, and a compact valid example.
|
||||||
|
- `docs/integrations/json-output.md` owns the optional logical file and
|
||||||
|
`index.json` descriptor; link to the evidence contract rather than repeating
|
||||||
|
its payload.
|
||||||
|
- `docs/operations.md` owns durable source-content sensitivity, permissions,
|
||||||
|
retention, and the possibility that selected lanes cover most of a
|
||||||
|
transcript.
|
||||||
|
- `docs/consumers/subprocess.md` explains discovery through the optional index
|
||||||
|
descriptor and requires consumers to treat `evidence_refs`, not expanded
|
||||||
|
context bounds, as citations.
|
||||||
|
- `docs/policy/architecture.md` records the generic typed evidence-projection
|
||||||
|
boundary and output ownership without adding D&D or wire-format detail.
|
||||||
|
- `docs/internal/pipeline.md` and `docs/internal/modules.md` describe the typed
|
||||||
|
evidence registry, preparation checks, reconstruction from serialized
|
||||||
|
normalize outputs, and output-stage ownership without restating public wire
|
||||||
|
fields.
|
||||||
|
|
||||||
|
Update only the smallest orientation links needed for discoverability. Do not
|
||||||
|
add a CLI flag, configuration environment override, or duplicate the complete
|
||||||
|
configuration outside `examples/`.
|
||||||
|
|
||||||
|
After all current-behavior documentation is accurate:
|
||||||
|
|
||||||
|
- set [the feature roadmap](evidence.md) status to `Implemented`;
|
||||||
|
- set this plan's status to `Completed`; and
|
||||||
|
- leave the integration and configuration documents, not either roadmap, as
|
||||||
|
the canonical implemented contract.
|
||||||
|
|
||||||
|
Final verification:
|
||||||
|
|
||||||
|
- `git diff --check`
|
||||||
|
- `go test ./...`
|
||||||
|
- `go vet ./...`
|
||||||
|
- `go build ./cmd/notarius`
|
||||||
|
- `go test -race ./internal/framework/evidencecontext ./internal/framework/pipeline ./internal/modules/generic/output/json ./internal/modules/dnd/... ./internal/modules/integration ./internal/cli`
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
|
||||||
|
None. The plan fixes the configuration shape and defaults, typed projection
|
||||||
|
boundary, preparation timing, durable schema and identities, range-union
|
||||||
|
algorithm, failure semantics, JSON discovery, D&D coverage, documentation
|
||||||
|
ownership, and test boundaries.
|
||||||
@@ -1,29 +0,0 @@
|
|||||||
# Future Work
|
|
||||||
|
|
||||||
Current Notarius behavior is documented in the canonical README, CLI,
|
|
||||||
configuration, operations, internal, and integration docs. This roadmap records
|
|
||||||
future work only.
|
|
||||||
|
|
||||||
## Candidate Product Work
|
|
||||||
|
|
||||||
- Additional input adapters, such as Markdown or note-export formats.
|
|
||||||
- Additional D&D extractors beyond spell casts.
|
|
||||||
- Cross-lane entity normalization.
|
|
||||||
- Cross-chunk semantic deduplication.
|
|
||||||
- Configurable validator chains with production validator modules.
|
|
||||||
- Multiple effective LLM profiles in one run.
|
|
||||||
- Parallel execution where it preserves deterministic manifests and diagnostics.
|
|
||||||
- Additional output encoders.
|
|
||||||
|
|
||||||
## Candidate Operational Work
|
|
||||||
|
|
||||||
- Packaged release artifacts for alpha distribution.
|
|
||||||
- A documented versioning and release process.
|
|
||||||
- Optional generated example output fixtures with a regeneration procedure.
|
|
||||||
- Additional diagnostics or reporting views if operator workflows need them.
|
|
||||||
|
|
||||||
## Non-Goals To Revisit Deliberately
|
|
||||||
|
|
||||||
- A general workflow language.
|
|
||||||
- Structural module selection through ad hoc run flags.
|
|
||||||
- Storing secrets in config files, diagnostics, manifests, or examples.
|
|
||||||
208
docs/roadmap/subprocess.md
Normal file
208
docs/roadmap/subprocess.md
Normal file
@@ -0,0 +1,208 @@
|
|||||||
|
# Subprocess Integration Contract
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
Implemented.
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
|
||||||
|
Make Notarius straightforward to invoke as a subprocess from an orchestrator
|
||||||
|
such as Narratio. A caller should be able to run a configured pipeline, discover
|
||||||
|
the published output bundle without parsing human prose or scanning a
|
||||||
|
directory, and hand selected structured artifacts to a later stage.
|
||||||
|
|
||||||
|
This work strengthens the public CLI boundary. It does not turn Notarius into a
|
||||||
|
Go library, embed Narratio-specific behavior, or change pipeline execution and
|
||||||
|
artifact semantics.
|
||||||
|
|
||||||
|
## Desired End State
|
||||||
|
|
||||||
|
A subprocess caller can:
|
||||||
|
|
||||||
|
1. validate a Notarius configuration and selected pipeline before execution;
|
||||||
|
2. invoke `notarius run` with explicit input, output-root, session, and
|
||||||
|
reference arguments;
|
||||||
|
3. request one versioned, machine-readable success result on standard output;
|
||||||
|
4. use that result to locate the published output bundle;
|
||||||
|
5. discover normalized lane payloads through the bundle's authoritative
|
||||||
|
`index.json`;
|
||||||
|
6. distinguish process failure from successful partial pipeline outcomes; and
|
||||||
|
7. record Notarius run provenance in its own manifest without depending on
|
||||||
|
internal packages, cache formats, debug formats, or human-readable messages.
|
||||||
|
|
||||||
|
The existing human-oriented command output remains the default for interactive
|
||||||
|
use.
|
||||||
|
|
||||||
|
## Machine-Readable Run Result
|
||||||
|
|
||||||
|
`notarius run` supports `--json`. On success, the flag makes standard output
|
||||||
|
contain exactly one JSON object followed by a newline. No human-oriented status
|
||||||
|
line is mixed into that stream.
|
||||||
|
|
||||||
|
The result uses the schema identity `notarius.run-result.v1` and contains:
|
||||||
|
|
||||||
|
| Field | Presence | Meaning |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| `schema_version` | Required | Exactly `notarius.run-result.v1`. |
|
||||||
|
| `run_id` | Required | The Notarius run identifier. |
|
||||||
|
| `pipeline_id` | Required | The effective pipeline identifier. |
|
||||||
|
| `output_directory` | Required | Absolute path to the successfully published output bundle. |
|
||||||
|
| `index_file` | Required for the production JSON output | Logical bundle path `index.json`. |
|
||||||
|
| `normalized_output_count` | Required | Number of final normalized lane outputs returned by the pipeline. |
|
||||||
|
| `rejected_output_count` | Required | Number of recorded rejected outputs. |
|
||||||
|
| `warning_count` | Required | Number of final run warnings returned by the pipeline. |
|
||||||
|
| `validation_status` | Required | The run manifest's final validation status without reinterpretation. |
|
||||||
|
| `debug_directory` | Optional | Absolute debug-bundle path when debug capture was requested and completed. |
|
||||||
|
|
||||||
|
An illustrative successful result is:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"schema_version": "notarius.run-result.v1",
|
||||||
|
"run_id": "run-1770000000000000000-0123456789abcdef0123456789abcdef",
|
||||||
|
"pipeline_id": "dnd-session",
|
||||||
|
"output_directory": "/srv/narratio/runs/session-7/notarius/run-1770000000000000000-0123456789abcdef0123456789abcdef",
|
||||||
|
"index_file": "index.json",
|
||||||
|
"normalized_output_count": 6,
|
||||||
|
"rejected_output_count": 2,
|
||||||
|
"warning_count": 1,
|
||||||
|
"validation_status": "approved"
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
The receipt is a discovery and summary document, not a duplicate output
|
||||||
|
envelope. It does not embed lane payloads, rejection entries, warnings, the run
|
||||||
|
manifest, or output-file contents. Consumers use `index_file` and the existing
|
||||||
|
published JSON output contract for those records.
|
||||||
|
|
||||||
|
The result contract must tolerate future additive optional fields. Any
|
||||||
|
incompatible field or semantic change requires a new run-result schema version.
|
||||||
|
|
||||||
|
## Stream, Publication, And Failure Semantics
|
||||||
|
|
||||||
|
Machine-readable output is emitted only after:
|
||||||
|
|
||||||
|
- the pipeline has completed without a framework error;
|
||||||
|
- all logical output files have been successfully published;
|
||||||
|
- requested debug terminal reporting has completed; and
|
||||||
|
- all result fields are known.
|
||||||
|
|
||||||
|
Writing or encoding the machine-readable result is part of successful command
|
||||||
|
completion. Failure to write it produces the existing runtime-failure exit
|
||||||
|
class.
|
||||||
|
|
||||||
|
With `--json`:
|
||||||
|
|
||||||
|
- successful stdout is exclusively the run-result JSON document;
|
||||||
|
- successful warnings remain on stderr under the existing CLI contract;
|
||||||
|
- syntax and runtime errors retain their existing exit statuses and stderr
|
||||||
|
diagnostics;
|
||||||
|
- consumers treat stdout as a valid result only when the process exits with
|
||||||
|
status 0; failures before result writing emit no result, while a failure
|
||||||
|
during the stdout write may leave incomplete bytes that must be ignored; and
|
||||||
|
- human-readable diagnostic wording is not promoted into a machine contract.
|
||||||
|
|
||||||
|
Without `--json`, current interactive stdout and stderr behavior remains
|
||||||
|
unchanged.
|
||||||
|
|
||||||
|
Successful runs may contain rejected outputs or omit some normalized lanes.
|
||||||
|
That remains a valid pipeline outcome. The run result reports counts, while
|
||||||
|
`index.json`, `rejected.json`, and `warnings.json` remain authoritative for
|
||||||
|
details. Notarius will not add a generic `--fail-on-rejection` policy as part
|
||||||
|
of this work.
|
||||||
|
|
||||||
|
## Output Discovery And Consumer Responsibilities
|
||||||
|
|
||||||
|
The production JSON encoder's `index.json` remains the authoritative mapping
|
||||||
|
from lane IDs to published payloads. A subprocess consumer should:
|
||||||
|
|
||||||
|
- resolve `index_file` beneath `output_directory` and reject path escape;
|
||||||
|
- locate expected outputs by `lane_id`, not by guessing filenames;
|
||||||
|
- check each selected descriptor's media type and schema identity;
|
||||||
|
- decode payloads according to their published integration contracts;
|
||||||
|
- decide which lanes are required or optional for its own later stages; and
|
||||||
|
- retain rejection, warning, and manifest files when they are needed for
|
||||||
|
provenance or review.
|
||||||
|
|
||||||
|
For Narratio, required report inputs and partial-success policy remain Narratio
|
||||||
|
stage configuration and orchestration concerns. Notarius does not acquire
|
||||||
|
knowledge of Narratio stages, manifests, workspace layout, publication policy,
|
||||||
|
or report formats.
|
||||||
|
|
||||||
|
## Invocation Guidance
|
||||||
|
|
||||||
|
The consumer documentation recommends that subprocess callers:
|
||||||
|
|
||||||
|
- use `notarius config validate --pipeline` as an optional preflight;
|
||||||
|
- pass explicit absolute paths for the input, configuration, output root, and
|
||||||
|
CLI-supplied references;
|
||||||
|
- use a stable, non-secret prompt session identifier when useful for provider
|
||||||
|
routing or caching;
|
||||||
|
- capture stdout and stderr separately;
|
||||||
|
- supply credentials through the configured environment mechanism rather than
|
||||||
|
command arguments or generated configuration containing secret values;
|
||||||
|
- place output, cache, debug, and subprocess logs under intentional
|
||||||
|
sensitivity and retention policies; and
|
||||||
|
- treat the Notarius manifest and run-result receipt as provenance while
|
||||||
|
leaving the caller's own manifest authoritative for its stage lifecycle.
|
||||||
|
|
||||||
|
Notarius configuration remains owned by Notarius. An orchestrator may select a
|
||||||
|
configuration and pass supported operational overrides, but should not
|
||||||
|
duplicate the complete Notarius configuration schema.
|
||||||
|
|
||||||
|
## Documentation End State
|
||||||
|
|
||||||
|
- `docs/cli.md` owns `run --json`, stream behavior, and exit semantics;
|
||||||
|
- a new `docs/integrations/run-result.md` owns the versioned run-result wire
|
||||||
|
contract and compatibility policy;
|
||||||
|
- `docs/integrations/json-output.md` remains the sole owner of output-bundle
|
||||||
|
discovery and lane publication;
|
||||||
|
- a new `docs/consumers/subprocess.md` provides the task-oriented invocation and
|
||||||
|
consumption workflow; and
|
||||||
|
- `docs/internal/cli.md` describes how the CLI constructs and emits the result
|
||||||
|
only after successful publication.
|
||||||
|
|
||||||
|
Other documents should link to these owners instead of repeating volatile
|
||||||
|
fields or command details.
|
||||||
|
|
||||||
|
## Acceptance Criteria
|
||||||
|
|
||||||
|
- An ordinary successful `run` retains its existing human-readable output.
|
||||||
|
- A successful `run --json` emits one valid `notarius.run-result.v1` document
|
||||||
|
and no human prose on stdout.
|
||||||
|
- Relative configured or overridden output and debug roots are reported as
|
||||||
|
absolute bundle paths.
|
||||||
|
- The receipt identifies the production JSON bundle entry point without
|
||||||
|
copying its lane descriptors or payloads.
|
||||||
|
- Warning-bearing and rejection-bearing runs remain successful and report
|
||||||
|
accurate counts.
|
||||||
|
- Syntax, configuration, provider, pipeline, publication, debug, and result
|
||||||
|
writing failures retain the correct nonzero exit class. Consumers are
|
||||||
|
explicitly required to ignore stdout from a nonzero invocation.
|
||||||
|
- The implementation does not expose internal Go types or couple generic CLI
|
||||||
|
code to D&D or Narratio concepts.
|
||||||
|
- Public and internal documentation assigns each new contract to one canonical
|
||||||
|
owner.
|
||||||
|
- Offline behavioral tests protect the structured-output contract, default
|
||||||
|
human behavior, absolute path reporting, stream separation, and failure to
|
||||||
|
serialize or write the success result without duplicating lower-level output
|
||||||
|
encoder tests.
|
||||||
|
|
||||||
|
## Out Of Scope
|
||||||
|
|
||||||
|
The following may be useful later but are not prerequisites for the Narratio
|
||||||
|
integration:
|
||||||
|
|
||||||
|
- a result-file flag in addition to machine-readable stdout;
|
||||||
|
- a JSON failure envelope or stable machine-readable error taxonomy;
|
||||||
|
- caller-supplied Notarius run IDs or exact output-bundle paths;
|
||||||
|
- a generic `--fail-on-rejection` or required-lane CLI policy;
|
||||||
|
- a public Go client package or importable Narratio adapter;
|
||||||
|
- Narratio stage, configuration, manifest, or report-generation changes;
|
||||||
|
- `notarius version --json`;
|
||||||
|
- installable or queryable artifact JSON Schemas;
|
||||||
|
- signal-aware CLI contexts and graceful SIGINT or SIGTERM handling;
|
||||||
|
- packaged release artifacts and a broader application-versioning policy.
|
||||||
|
|
||||||
|
These items should be promoted only in response to a demonstrated integration
|
||||||
|
need rather than bundled into the initial subprocess contract.
|
||||||
@@ -1,215 +0,0 @@
|
|||||||
# Troubleshooting
|
|
||||||
|
|
||||||
This guide maps common implemented failure modes to inspection steps and fixes.
|
|
||||||
For command syntax, see [CLI Reference](cli.md). For YAML fields and
|
|
||||||
environment overrides, see [Configuration](config.md). For output and
|
|
||||||
diagnostics layout, see [Operations](operations.md).
|
|
||||||
|
|
||||||
## Config File Not Found
|
|
||||||
|
|
||||||
Symptom:
|
|
||||||
|
|
||||||
```text
|
|
||||||
notarius: config file not found; pass --config or set NOTARIUS_CONFIG
|
|
||||||
```
|
|
||||||
|
|
||||||
Fix:
|
|
||||||
|
|
||||||
- Pass `--config path/to/config.yml`.
|
|
||||||
- Or set `NOTARIUS_CONFIG` to a readable file.
|
|
||||||
- Or install a config at `/usr/local/etc/notarius/config.yml`.
|
|
||||||
|
|
||||||
If the message says the config path is a directory or is not available, correct
|
|
||||||
the path or file permissions.
|
|
||||||
|
|
||||||
## Unsupported Or Invalid Config
|
|
||||||
|
|
||||||
Symptoms include:
|
|
||||||
|
|
||||||
- `unsupported config version`
|
|
||||||
- `config version is required`
|
|
||||||
- `field <name> not found`
|
|
||||||
- `total LLM concurrency must be greater than zero`
|
|
||||||
- `diagnostics retention "<value>" is not supported`
|
|
||||||
|
|
||||||
Fix:
|
|
||||||
|
|
||||||
- Use `version: 1`.
|
|
||||||
- Remove unknown YAML fields.
|
|
||||||
- Validate with:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
go run ./cmd/notarius config validate --config path/to/config.yml
|
|
||||||
```
|
|
||||||
|
|
||||||
## Unknown Pipeline
|
|
||||||
|
|
||||||
Symptom:
|
|
||||||
|
|
||||||
```text
|
|
||||||
notarius: pipeline "..." is not configured
|
|
||||||
```
|
|
||||||
|
|
||||||
Fix:
|
|
||||||
|
|
||||||
- List configured pipeline IDs:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
go run ./cmd/notarius pipelines list --config path/to/config.yml
|
|
||||||
```
|
|
||||||
|
|
||||||
- Use one of those IDs in `notarius run <pipeline-id>`.
|
|
||||||
- Check indentation under the top-level `pipelines` map.
|
|
||||||
|
|
||||||
## Unknown Or Incompatible Module
|
|
||||||
|
|
||||||
Symptoms mention a module key, pipeline slot, lane, capability, or `not
|
|
||||||
registered`.
|
|
||||||
|
|
||||||
Fix:
|
|
||||||
|
|
||||||
- Validate the pipeline against the production module catalog:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
go run ./cmd/notarius config validate \
|
|
||||||
--config path/to/config.yml \
|
|
||||||
--pipeline dnd-session
|
|
||||||
```
|
|
||||||
|
|
||||||
- Use only implemented production module keys listed in
|
|
||||||
[Configuration](config.md#implemented-production-modules).
|
|
||||||
- Check that artifact lanes include an `extract` binding.
|
|
||||||
|
|
||||||
## Invalid `--only`
|
|
||||||
|
|
||||||
Symptoms include:
|
|
||||||
|
|
||||||
- `--only must contain comma-separated non-empty artifact lane IDs`
|
|
||||||
- `--only requires --pipeline`
|
|
||||||
- `selected artifact lane`
|
|
||||||
|
|
||||||
Fix:
|
|
||||||
|
|
||||||
- Use comma-separated lane IDs with no empty entries:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
go run ./cmd/notarius run dnd-session \
|
|
||||||
--config path/to/config.yml \
|
|
||||||
--input path/to/input.json \
|
|
||||||
--only spells
|
|
||||||
```
|
|
||||||
|
|
||||||
- For `config validate`, include `--pipeline` when using `--only`.
|
|
||||||
- Confirm the lane ID exists under `pipelines.<id>.artifacts`.
|
|
||||||
|
|
||||||
## Seriatim Input Validation Failure
|
|
||||||
|
|
||||||
Symptoms include `seriatim input`, `parse JSON`, `segments must not be empty`,
|
|
||||||
or validation errors naming a segment field.
|
|
||||||
|
|
||||||
Fix:
|
|
||||||
|
|
||||||
- Compare the input to
|
|
||||||
[examples/seriatim-minimal-transcript.json](../examples/seriatim-minimal-transcript.json).
|
|
||||||
- Ensure the JSON has a `metadata` object and a non-empty `segments` array.
|
|
||||||
- Each segment needs a non-empty `id`, non-empty `speaker`, non-empty `text`,
|
|
||||||
non-negative numeric `start`, and non-negative numeric `end`.
|
|
||||||
- Segment IDs must be unique and must not contain leading or trailing
|
|
||||||
whitespace.
|
|
||||||
- `end` must be greater than or equal to `start`.
|
|
||||||
|
|
||||||
## Missing LLM Base URL Or Model
|
|
||||||
|
|
||||||
Symptoms include:
|
|
||||||
|
|
||||||
- `LLM profile "default" base URL must not be empty`
|
|
||||||
- `LLM profile "default" model must not be empty`
|
|
||||||
- `base URL must be valid`
|
|
||||||
|
|
||||||
Fix:
|
|
||||||
|
|
||||||
- Set `base_url` and `model` in `llm_profiles.default`.
|
|
||||||
- Or set `NOTARIUS_LLM_DEFAULT_BASE_URL` and
|
|
||||||
`NOTARIUS_LLM_DEFAULT_MODEL`.
|
|
||||||
- If a profile needs authentication, set `api_key_env` in YAML or set
|
|
||||||
`NOTARIUS_LLM_DEFAULT_API_KEY`.
|
|
||||||
|
|
||||||
## LLM Profile Override Failure
|
|
||||||
|
|
||||||
Symptom:
|
|
||||||
|
|
||||||
```text
|
|
||||||
notarius: LLM profile override "..." is not configured
|
|
||||||
```
|
|
||||||
|
|
||||||
Fix:
|
|
||||||
|
|
||||||
- Add the profile under `llm_profiles`.
|
|
||||||
- Or use an existing profile ID with `--llm-profile`.
|
|
||||||
|
|
||||||
Current runs require exactly one distinct effective LLM profile. If a pipeline
|
|
||||||
uses several profiles, run with `--llm-profile <id>` or align the bindings in
|
|
||||||
configuration.
|
|
||||||
|
|
||||||
## Provider HTTP Or Response Failure
|
|
||||||
|
|
||||||
Symptoms include:
|
|
||||||
|
|
||||||
- `provider request failed`
|
|
||||||
- `provider returned status 400`
|
|
||||||
- `provider returned status 403`
|
|
||||||
- `provider response missing choices`
|
|
||||||
- `provider response assistant message content is not valid JSON`
|
|
||||||
- `decode structured output`
|
|
||||||
|
|
||||||
Fix:
|
|
||||||
|
|
||||||
- Confirm the `base_url` points to an OpenAI-compatible endpoint root. Notarius
|
|
||||||
posts to `<base_url>/chat/completions`.
|
|
||||||
- Check `model` and provider credentials.
|
|
||||||
- Inspect the retained diagnostics `error.log`.
|
|
||||||
- For 400 and 403 responses, fix the request configuration or credentials.
|
|
||||||
- For 429 and 5xx responses, the client retries according to `max_retries`; if
|
|
||||||
the failure persists, inspect the provider response and adjust capacity,
|
|
||||||
credentials, or model settings.
|
|
||||||
- The assistant message content must decode as JSON matching the extractor's
|
|
||||||
structured response schema.
|
|
||||||
|
|
||||||
Provider error messages are redacted for configured API key values.
|
|
||||||
|
|
||||||
## Output Write Failure
|
|
||||||
|
|
||||||
Symptoms include:
|
|
||||||
|
|
||||||
- `create output directory`
|
|
||||||
- `write output file`
|
|
||||||
- `output file name must`
|
|
||||||
|
|
||||||
Fix:
|
|
||||||
|
|
||||||
- Ensure `--output-dir` points to a directory path or a path that can be
|
|
||||||
created.
|
|
||||||
- Check filesystem permissions and available disk space.
|
|
||||||
- If diagnostics were retained, inspect `run-report.json`, `run-manifest.json`,
|
|
||||||
and `error.log`.
|
|
||||||
|
|
||||||
The CLI rejects unsafe logical output paths before writing files.
|
|
||||||
|
|
||||||
## Diagnostics Directory Surprise
|
|
||||||
|
|
||||||
Symptom: the diagnostics directory is missing after a successful run.
|
|
||||||
|
|
||||||
Fix:
|
|
||||||
|
|
||||||
- Check `diagnostics.retention`.
|
|
||||||
- With `auto`, successful runs without warnings are removed.
|
|
||||||
- Use `diagnostics.retention: always` when every diagnostics run directory
|
|
||||||
should be kept.
|
|
||||||
- Use `--diagnostics-dir` to override the configured work directory for a run.
|
|
||||||
|
|
||||||
Symptom: diagnostics exist even with `retention: never`.
|
|
||||||
|
|
||||||
Explanation:
|
|
||||||
|
|
||||||
- Failed runs are retained so that `error.log` and available context can be
|
|
||||||
inspected.
|
|
||||||
85
examples/dnd-complete-transcript.json
Normal file
85
examples/dnd-complete-transcript.json
Normal file
@@ -0,0 +1,85 @@
|
|||||||
|
{
|
||||||
|
"metadata": {
|
||||||
|
"id": "session-ravenfall",
|
||||||
|
"title": "The Ravenfall Watchtower"
|
||||||
|
},
|
||||||
|
"segments": [
|
||||||
|
{
|
||||||
|
"id": 1,
|
||||||
|
"start": 0,
|
||||||
|
"end": 14,
|
||||||
|
"speaker": "DM",
|
||||||
|
"text": "Recap: last session, the party learned that Elder Rowan vanished near the Ravenfall watchtower."
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 2,
|
||||||
|
"start": 14,
|
||||||
|
"end": 25,
|
||||||
|
"speaker": "Player",
|
||||||
|
"text": "Out of character, we agree to investigate the watchtower before the next game."
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 3,
|
||||||
|
"start": 25,
|
||||||
|
"end": 39,
|
||||||
|
"speaker": "DM",
|
||||||
|
"text": "Aria and Borin arrive at the ruined Ravenfall watchtower as dusk settles over the road."
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 4,
|
||||||
|
"start": 39,
|
||||||
|
"end": 55,
|
||||||
|
"speaker": "Mira Thorn",
|
||||||
|
"text": "Mira Thorn steps from the doorway and says, \"Elder Rowan warned me that Kesh would return for the relic.\""
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 5,
|
||||||
|
"start": 55,
|
||||||
|
"end": 70,
|
||||||
|
"speaker": "DM",
|
||||||
|
"text": "Mira leads the party to a hidden cache. The party discovers a moonblade and acquires 20 silver pieces."
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 6,
|
||||||
|
"start": 70,
|
||||||
|
"end": 83,
|
||||||
|
"speaker": "Aria",
|
||||||
|
"text": "Aria hands her healing potion to Borin so he can carry it into the tower."
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 7,
|
||||||
|
"start": 83,
|
||||||
|
"end": 96,
|
||||||
|
"speaker": "DM",
|
||||||
|
"text": "Kesh, the goblin captain, orders the raiders to attack. Roll initiative."
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 8,
|
||||||
|
"start": 96,
|
||||||
|
"end": 110,
|
||||||
|
"speaker": "DM",
|
||||||
|
"text": "On Kesh's turn, he strikes Borin with his scimitar. Borin drinks the healing potion on his turn."
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 9,
|
||||||
|
"start": 110,
|
||||||
|
"end": 124,
|
||||||
|
"speaker": "Aria",
|
||||||
|
"text": "Aria casts Cure Wounds on Borin, then invokes Aegis of Emberfall as Kesh closes in."
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 10,
|
||||||
|
"start": 124,
|
||||||
|
"end": 137,
|
||||||
|
"speaker": "DM",
|
||||||
|
"text": "Kesh casts Shield as a reaction against Borin's counterattack, but the party drives the raiders away."
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 11,
|
||||||
|
"start": 137,
|
||||||
|
"end": 150,
|
||||||
|
"speaker": "DM",
|
||||||
|
"text": "After the battle, Aria pays 5 silver pieces to repair the watchtower gate."
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
101
examples/dnd-complete.config.yml
Normal file
101
examples/dnd-complete.config.yml
Normal file
@@ -0,0 +1,101 @@
|
|||||||
|
version: 3
|
||||||
|
concurrency:
|
||||||
|
total_llm: 2
|
||||||
|
stage_workers:
|
||||||
|
extract: 2
|
||||||
|
output:
|
||||||
|
directory: ./notarius-output
|
||||||
|
cache:
|
||||||
|
chunk_plans:
|
||||||
|
mode: auto
|
||||||
|
directory: ./notarius-cache/chunk-plans
|
||||||
|
checkpoints:
|
||||||
|
enabled: true
|
||||||
|
directory: ./notarius-cache/checkpoints
|
||||||
|
debug:
|
||||||
|
directory: ./notarius-debug
|
||||||
|
pipelines:
|
||||||
|
dnd-session:
|
||||||
|
input: seriatim
|
||||||
|
# Stable campaign context is shared by every module that accepts these slots.
|
||||||
|
references:
|
||||||
|
party: ./dnd-party.txt
|
||||||
|
glossary: ./dnd-glossary.txt
|
||||||
|
chunk:
|
||||||
|
module: dnd/scenes
|
||||||
|
retries: 2
|
||||||
|
output:
|
||||||
|
module: json
|
||||||
|
options:
|
||||||
|
include_chunk_map: true
|
||||||
|
evidence_context:
|
||||||
|
enabled: true
|
||||||
|
window_units: 3
|
||||||
|
lanes:
|
||||||
|
- item-events
|
||||||
|
- npcs
|
||||||
|
- spells
|
||||||
|
- combat-turns
|
||||||
|
- npc-interactions
|
||||||
|
steps:
|
||||||
|
# Establish session-wide reference artifacts alongside independent item events.
|
||||||
|
- id: describe-session
|
||||||
|
artifacts:
|
||||||
|
item-events:
|
||||||
|
extract:
|
||||||
|
module: dnd/item-events
|
||||||
|
retries: 2
|
||||||
|
merge: appendorder
|
||||||
|
normalize: dnd/item-events
|
||||||
|
npcs:
|
||||||
|
extract:
|
||||||
|
module: dnd/npcs
|
||||||
|
retries: 2
|
||||||
|
merge: appendorder
|
||||||
|
normalize:
|
||||||
|
module: dnd/npcs
|
||||||
|
llm_profile: gemini-2-flash
|
||||||
|
retries: 2
|
||||||
|
scene-descriptions:
|
||||||
|
extract:
|
||||||
|
module: dnd/scene-descriptions
|
||||||
|
retries: 2
|
||||||
|
merge: appendorder
|
||||||
|
normalize: dnd/scene-descriptions
|
||||||
|
- id: extract-events
|
||||||
|
# Accepted NPC grounding and scene-description eligibility artifacts are
|
||||||
|
# supplied in memory to their compatible consumers in this step.
|
||||||
|
references:
|
||||||
|
npcs:
|
||||||
|
artifact:
|
||||||
|
step: describe-session
|
||||||
|
lane: npcs
|
||||||
|
scene_descriptions:
|
||||||
|
artifact:
|
||||||
|
step: describe-session
|
||||||
|
lane: scene-descriptions
|
||||||
|
artifacts:
|
||||||
|
spells:
|
||||||
|
extract:
|
||||||
|
module: dnd/spells
|
||||||
|
retries: 2
|
||||||
|
references:
|
||||||
|
spell_catalog: ./dnd-spell-catalog.json
|
||||||
|
merge: appendorder
|
||||||
|
# Stage-local file references are intentionally bound at each stage.
|
||||||
|
normalize:
|
||||||
|
module: dnd/spells
|
||||||
|
references:
|
||||||
|
spell_catalog: ./dnd-spell-catalog.json
|
||||||
|
combat-turns:
|
||||||
|
extract:
|
||||||
|
module: dnd/combat-turns
|
||||||
|
retries: 2
|
||||||
|
merge: appendorder
|
||||||
|
normalize: dnd/combat-turns
|
||||||
|
npc-interactions:
|
||||||
|
extract:
|
||||||
|
module: dnd/npc-interactions
|
||||||
|
retries: 2
|
||||||
|
merge: appendorder
|
||||||
|
normalize: dnd/npc-interactions
|
||||||
8
examples/dnd-glossary.txt
Normal file
8
examples/dnd-glossary.txt
Normal file
@@ -0,0 +1,8 @@
|
|||||||
|
Ravenfall watchtower: a ruined watchtower near the party's current route.
|
||||||
|
Mira Thorn: the watchtower's keeper.
|
||||||
|
Elder Rowan: a missing local scholar.
|
||||||
|
Kesh: a goblin captain leading raiders.
|
||||||
|
Moonblade: a blade found in the watchtower's hidden cache.
|
||||||
|
Cure Wounds: a healing spell.
|
||||||
|
Shield: a defensive reaction spell.
|
||||||
|
Aegis of Emberfall: a campaign spell recorded in the supplied catalog overlay.
|
||||||
8
examples/dnd-minimal.config.yml
Normal file
8
examples/dnd-minimal.config.yml
Normal file
@@ -0,0 +1,8 @@
|
|||||||
|
version: 3
|
||||||
|
pipelines:
|
||||||
|
dnd-session:
|
||||||
|
input: seriatim
|
||||||
|
artifacts:
|
||||||
|
spells:
|
||||||
|
extract: dnd/spells
|
||||||
|
normalize: dnd/spells
|
||||||
2
examples/dnd-party.txt
Normal file
2
examples/dnd-party.txt
Normal file
@@ -0,0 +1,2 @@
|
|||||||
|
Aria: party cleric and recurring healer.
|
||||||
|
Borin: fighter ally.
|
||||||
21
examples/dnd-spell-catalog.json
Normal file
21
examples/dnd-spell-catalog.json
Normal file
@@ -0,0 +1,21 @@
|
|||||||
|
{
|
||||||
|
"schema_version": "notarius.dnd.spell-catalog-overlay.v1",
|
||||||
|
"catalogs": [
|
||||||
|
{
|
||||||
|
"id": "notarius.example-campaign",
|
||||||
|
"ruleset": "dnd-5e-2014",
|
||||||
|
"source": {
|
||||||
|
"title": "Notarius example campaign spell names",
|
||||||
|
"version": "1",
|
||||||
|
"url": "",
|
||||||
|
"license": ""
|
||||||
|
},
|
||||||
|
"spells": [
|
||||||
|
{
|
||||||
|
"name": "Aegis of Emberfall",
|
||||||
|
"aliases": ["Emberfall Aegis"]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -1,16 +0,0 @@
|
|||||||
version: 1
|
|
||||||
llm_profiles:
|
|
||||||
default:
|
|
||||||
provider: openai-compatible
|
|
||||||
base_url: http://127.0.0.1:1
|
|
||||||
model: fake-model
|
|
||||||
pipelines:
|
|
||||||
dnd-session:
|
|
||||||
input: seriatim
|
|
||||||
chunk:
|
|
||||||
module: generic
|
|
||||||
options:
|
|
||||||
max_units: 50
|
|
||||||
artifacts:
|
|
||||||
spells:
|
|
||||||
extract: dnd/spells
|
|
||||||
@@ -5,14 +5,14 @@
|
|||||||
},
|
},
|
||||||
"segments": [
|
"segments": [
|
||||||
{
|
{
|
||||||
"id": "seg-001",
|
"id": 1,
|
||||||
"start": 0,
|
"start": 0,
|
||||||
"end": 4,
|
"end": 4,
|
||||||
"speaker": "Aria",
|
"speaker": "Aria",
|
||||||
"text": "Aria raises her holy symbol and casts Cure Wounds."
|
"text": "Aria raises her holy symbol and casts Cure Wounds."
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"id": "seg-002",
|
"id": 2,
|
||||||
"start": 4,
|
"start": 4,
|
||||||
"end": 8,
|
"end": 8,
|
||||||
"speaker": "DM",
|
"speaker": "DM",
|
||||||
|
|||||||
10
go.mod
10
go.mod
@@ -1,5 +1,11 @@
|
|||||||
module gitea.maximumdirect.net/eric/notarius
|
module gitea.maximumdirect.net/eric/notarius
|
||||||
|
|
||||||
go 1.24.0
|
go 1.25.5
|
||||||
|
|
||||||
require gopkg.in/yaml.v3 v3.0.1
|
require (
|
||||||
|
gitea.maximumdirect.net/eric/scriptorium v0.11.1
|
||||||
|
github.com/santhosh-tekuri/jsonschema/v6 v6.0.2
|
||||||
|
gopkg.in/yaml.v3 v3.0.1
|
||||||
|
)
|
||||||
|
|
||||||
|
require golang.org/x/text v0.40.0
|
||||||
|
|||||||
12
go.sum
12
go.sum
@@ -1,3 +1,15 @@
|
|||||||
|
gitea.maximumdirect.net/eric/scriptorium v0.11.0 h1:rjvbt9FTaWHxYlHq7QlUzmMVUt3QdbTmeCkmH81N//o=
|
||||||
|
gitea.maximumdirect.net/eric/scriptorium v0.11.0/go.mod h1:FQ5lEuNxmrQyNgIomkpZdxvfTC0jWjbXYuq3tbJWF64=
|
||||||
|
gitea.maximumdirect.net/eric/scriptorium v0.11.1 h1:zBKtB3+fP8FcHGI8DJD99CiTL6crAGitBhWtE+xYJHc=
|
||||||
|
gitea.maximumdirect.net/eric/scriptorium v0.11.1/go.mod h1:FQ5lEuNxmrQyNgIomkpZdxvfTC0jWjbXYuq3tbJWF64=
|
||||||
|
github.com/dlclark/regexp2 v1.11.0 h1:G/nrcoOa7ZXlpoa/91N3X7mM3r8eIlMBBJZvsz/mxKI=
|
||||||
|
github.com/dlclark/regexp2 v1.11.0/go.mod h1:DHkYz0B9wPfa6wondMfaivmHpzrQ3v9q8cnmRbL6yW8=
|
||||||
|
github.com/santhosh-tekuri/jsonschema/v6 v6.0.2 h1:KRzFb2m7YtdldCEkzs6KqmJw4nqEVZGK7IN2kJkjTuQ=
|
||||||
|
github.com/santhosh-tekuri/jsonschema/v6 v6.0.2/go.mod h1:JXeL+ps8p7/KNMjDQk3TCwPpBy0wYklyWTfbkIzdIFU=
|
||||||
|
golang.org/x/text v0.14.0 h1:ScX5w1eTa3QqT8oi6+ziP7dTV1S2+ALU0bI+0zXKWiQ=
|
||||||
|
golang.org/x/text v0.14.0/go.mod h1:18ZOQIKpY8NJVqYksKHtTdi31H5itFRjB5/qKTNYzSU=
|
||||||
|
golang.org/x/text v0.40.0 h1:Ub2Z6/xjgF1WrYQz2nuITOEegKFtiIy+rieRJ5lHZKs=
|
||||||
|
golang.org/x/text v0.40.0/go.mod h1:hpnzDAfGV753zIKo+wk3u1bVKCGPbrnF7+7LBF/UHVY=
|
||||||
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM=
|
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM=
|
||||||
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
|
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
|
||||||
gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=
|
gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=
|
||||||
|
|||||||
314
internal/cli/assembled_spell_pipeline_contract_test.go
Normal file
314
internal/cli/assembled_spell_pipeline_contract_test.go
Normal file
@@ -0,0 +1,314 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
|
"reflect"
|
||||||
|
"sort"
|
||||||
|
"strings"
|
||||||
|
"sync"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||||
|
spellnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/spells"
|
||||||
|
spellcatalog "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/spells/catalog"
|
||||||
|
)
|
||||||
|
|
||||||
|
const assembledSpellExtractorKey = "test/dnd/spell-casts"
|
||||||
|
|
||||||
|
func TestAssembledSpellPipelineNormalizesMergedCasts(t *testing.T) {
|
||||||
|
registries, resolved, extractor := assembledSpellPipeline(t, assembledSpellPipelineOptions{})
|
||||||
|
prepared, err := pipeline.Prepare(resolved, registries, pipeline.ModuleDependencies{})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Prepare() error = %v, want nil", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
output, err := pipeline.New().Run(context.Background(), pipeline.RunInput{
|
||||||
|
Prepared: prepared,
|
||||||
|
RawInput: readRepositoryFile(t, "examples", "seriatim-minimal-transcript.json"),
|
||||||
|
ChunkCacheMode: pipeline.ChunkCacheBypass,
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Run() error = %v, want nil", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
chunkIndexes := extractor.chunkIndexesSnapshot()
|
||||||
|
sort.Ints(chunkIndexes)
|
||||||
|
if !reflect.DeepEqual(chunkIndexes, []int{0, 1}) {
|
||||||
|
t.Fatalf("extractor chunk indexes = %#v, want two chunk-boundary calls", chunkIndexes)
|
||||||
|
}
|
||||||
|
if output.Manifest.ValidationStatus != "approved" || len(output.Rejected) != 0 || len(output.NormalizeOutputs) != 1 {
|
||||||
|
t.Fatalf("run output = %#v, want approved normalized output without rejections", output)
|
||||||
|
}
|
||||||
|
if output.NormalizeOutputs[0].NormalizerKey != spellnormalize.Key {
|
||||||
|
t.Fatalf("normalized output module = %q, want %q", output.NormalizeOutputs[0].NormalizerKey, spellnormalize.Key)
|
||||||
|
}
|
||||||
|
|
||||||
|
var normalized dnd.SpellList
|
||||||
|
if err := json.Unmarshal(output.NormalizeOutputs[0].Artifact.Content, &normalized); err != nil {
|
||||||
|
t.Fatalf("decode normalized output: %v", err)
|
||||||
|
}
|
||||||
|
if len(normalized.SpellCasts) != 2 {
|
||||||
|
t.Fatalf("normalized casts = %#v, want collapsed duplicate plus distinct evidence", normalized.SpellCasts)
|
||||||
|
}
|
||||||
|
first, distinct := normalized.SpellCasts[0], normalized.SpellCasts[1]
|
||||||
|
if first.Spell != "Cure Wounds" || first.Caster != " Aria \t" {
|
||||||
|
t.Fatalf("retained cast = %#v, want canonical spell with first occurrence caster", first)
|
||||||
|
}
|
||||||
|
if !reflect.DeepEqual(first.SourceRefs, []source.SourceRef{{SourceID: "session-alpha", StartUnitID: 1, EndUnitID: 1}, {SourceID: "session-alpha", StartUnitID: 2, EndUnitID: 2}}) {
|
||||||
|
t.Fatalf("retained refs = %#v, want sorted complete evidence", first.SourceRefs)
|
||||||
|
}
|
||||||
|
if distinct.Spell != "Cure Wounds" || distinct.Caster != "aria" || !reflect.DeepEqual(distinct.SourceRefs, []source.SourceRef{{SourceID: "session-alpha", StartUnitID: 2, EndUnitID: 2}}) {
|
||||||
|
t.Fatalf("distinct cast = %#v, want separate evidence event", distinct)
|
||||||
|
}
|
||||||
|
|
||||||
|
wantWarningReasons := []string{
|
||||||
|
spellnormalize.ReasonCodeSpellNameCanonicalized,
|
||||||
|
spellnormalize.ReasonCodeSourceReferencesNormalized,
|
||||||
|
spellnormalize.ReasonCodeDuplicateSpellCastCollapsed,
|
||||||
|
"spell_not_near_source",
|
||||||
|
}
|
||||||
|
gotWarningReasons := make([]string, len(output.Warnings))
|
||||||
|
for index, warning := range output.Warnings {
|
||||||
|
gotWarningReasons[index] = warning.ReasonCode
|
||||||
|
}
|
||||||
|
if !reflect.DeepEqual(gotWarningReasons, wantWarningReasons) {
|
||||||
|
t.Fatalf("warnings = %#v, want deterministic normalize and validation warnings", output.Warnings)
|
||||||
|
}
|
||||||
|
if output.Warnings[2].Scope != "spell_casts[0]" || !strings.Contains(output.Warnings[2].Message, "retained input index 0") || !strings.Contains(output.Warnings[2].Message, "removed input indices [1]") {
|
||||||
|
t.Fatalf("duplicate warning = %#v, want retained and removed merged indices", output.Warnings[2])
|
||||||
|
}
|
||||||
|
|
||||||
|
warningsFile := decodeAssembledOutput[struct {
|
||||||
|
Warnings []contracts.Warning `json:"warnings"`
|
||||||
|
}](t, output.OutputFiles, "warnings.json")
|
||||||
|
if !reflect.DeepEqual(warningsFile.Warnings, output.Warnings) {
|
||||||
|
t.Fatalf("warnings file = %#v, run warnings = %#v, want manifest output path to preserve warnings", warningsFile.Warnings, output.Warnings)
|
||||||
|
}
|
||||||
|
manifest := decodeAssembledOutput[artifacts.RunManifest](t, output.OutputFiles, "manifest.json")
|
||||||
|
if len(manifest.ArtifactLanes) != 1 || manifest.ArtifactLanes[0].Normalizer != spellnormalize.Key {
|
||||||
|
t.Fatalf("manifest lanes = %#v, want assembled spell normalizer", manifest.ArtifactLanes)
|
||||||
|
}
|
||||||
|
normalizerMetadata, ok := manifest.ArtifactLanes[0].Metadata["normalizer"].(map[string]any)
|
||||||
|
_, hasOverlayIDs := normalizerMetadata["catalog_overlay_ids"]
|
||||||
|
if !ok || normalizerMetadata["catalog_base_id"] != spellcatalog.SRD5E2014ID || !strings.HasPrefix(stringValue(normalizerMetadata["catalog_digest"]), "sha256:") || !hasOverlayIDs {
|
||||||
|
t.Fatalf("normalizer manifest metadata = %#v, want base ID, digest, and overlay IDs", manifest.ArtifactLanes[0].Metadata)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAssembledSpellPipelineHonorsNormalizeValidatorOverride(t *testing.T) {
|
||||||
|
registries, resolved, _ := assembledSpellPipeline(t, assembledSpellPipelineOptions{normalizeValidatorOverride: true})
|
||||||
|
var normalizeChain *pipeline.ResolvedValidatorChain
|
||||||
|
for index := range resolved.ValidatorChains {
|
||||||
|
chain := &resolved.ValidatorChains[index]
|
||||||
|
if chain.Stage == pipeline.StageNormalize && chain.ModuleKey == spellnormalize.Key && chain.LaneID == "spells" {
|
||||||
|
normalizeChain = chain
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if normalizeChain == nil || len(normalizeChain.Validators) != 1 || normalizeChain.Validators[0].Binding.Module != "generic/always_accept" {
|
||||||
|
t.Fatalf("normalize validator chain = %#v, want explicit always-accept override", normalizeChain)
|
||||||
|
}
|
||||||
|
|
||||||
|
prepared, err := pipeline.Prepare(resolved, registries, pipeline.ModuleDependencies{})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Prepare() error = %v, want nil", err)
|
||||||
|
}
|
||||||
|
output, err := pipeline.New().Run(context.Background(), pipeline.RunInput{
|
||||||
|
Prepared: prepared,
|
||||||
|
RawInput: readRepositoryFile(t, "examples", "seriatim-minimal-transcript.json"),
|
||||||
|
ChunkCacheMode: pipeline.ChunkCacheBypass,
|
||||||
|
})
|
||||||
|
if err != nil || output.Manifest.ValidationStatus != "approved" || len(output.Rejected) != 0 || len(output.NormalizeOutputs) != 1 {
|
||||||
|
t.Fatalf("Run() error = %v output = %#v, want approved override run", err, output)
|
||||||
|
}
|
||||||
|
for _, warning := range output.Warnings {
|
||||||
|
if warning.ReasonCode == "spell_not_near_source" {
|
||||||
|
t.Fatalf("warnings = %#v, want explicit validator override to replace default relatedness chain", output.Warnings)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAssembledSpellPipelineRejectsUnknownSpellWithoutPromotingAttemptWarning(t *testing.T) {
|
||||||
|
registries, resolved, _ := assembledSpellPipeline(t, assembledSpellPipelineOptions{unknownSpell: true})
|
||||||
|
prepared, err := pipeline.Prepare(resolved, registries, pipeline.ModuleDependencies{})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Prepare() error = %v, want nil", err)
|
||||||
|
}
|
||||||
|
output, err := pipeline.New().Run(context.Background(), pipeline.RunInput{
|
||||||
|
Prepared: prepared,
|
||||||
|
RawInput: readRepositoryFile(t, "examples", "seriatim-minimal-transcript.json"),
|
||||||
|
ChunkCacheMode: pipeline.ChunkCacheBypass,
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Run() error = %v, want nil", err)
|
||||||
|
}
|
||||||
|
if output.Manifest.ValidationStatus != "rejected" || len(output.NormalizeOutputs) != 0 || len(output.Rejected) != 1 {
|
||||||
|
t.Fatalf("run output = %#v, want one rejected normalize candidate and no normalized output", output)
|
||||||
|
}
|
||||||
|
rejection := output.Rejected[0]
|
||||||
|
if rejection.Stage != string(pipeline.StageNormalize) || rejection.LaneID != "spells" || rejection.ModuleKey != spellnormalize.Key || rejection.ValidatorName != "extract/dnd/spells/catalog" || rejection.ReasonCode != "unknown_spell" {
|
||||||
|
t.Fatalf("rejection = %#v, want durable normalize catalog rejection", rejection)
|
||||||
|
}
|
||||||
|
rejectedFile := decodeAssembledOutput[struct {
|
||||||
|
Rejected []contracts.RejectedOutput `json:"rejected"`
|
||||||
|
}](t, output.OutputFiles, "rejected.json")
|
||||||
|
if !reflect.DeepEqual(rejectedFile.Rejected, output.Rejected) {
|
||||||
|
t.Fatalf("rejected file = %#v, run rejections = %#v, want durable rejection diagnostic", rejectedFile.Rejected, output.Rejected)
|
||||||
|
}
|
||||||
|
for _, warning := range output.Warnings {
|
||||||
|
if warning.ReasonCode == spellnormalize.ReasonCodeSpellNameUnresolved {
|
||||||
|
t.Fatalf("warnings = %#v, want rejected-attempt warning to remain non-durable", output.Warnings)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAssembledSpellPipelinePromotesUnknownSpellWarningWhenOverrideAccepts(t *testing.T) {
|
||||||
|
registries, resolved, _ := assembledSpellPipeline(t, assembledSpellPipelineOptions{normalizeValidatorOverride: true, unknownSpell: true})
|
||||||
|
prepared, err := pipeline.Prepare(resolved, registries, pipeline.ModuleDependencies{})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Prepare() error = %v, want nil", err)
|
||||||
|
}
|
||||||
|
output, err := pipeline.New().Run(context.Background(), pipeline.RunInput{
|
||||||
|
Prepared: prepared,
|
||||||
|
RawInput: readRepositoryFile(t, "examples", "seriatim-minimal-transcript.json"),
|
||||||
|
ChunkCacheMode: pipeline.ChunkCacheBypass,
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Run() error = %v, want nil", err)
|
||||||
|
}
|
||||||
|
if output.Manifest.ValidationStatus != "approved" || len(output.Rejected) != 0 || len(output.NormalizeOutputs) != 1 {
|
||||||
|
t.Fatalf("run output = %#v, want accepted unknown spell with explicit validator override", output)
|
||||||
|
}
|
||||||
|
var normalized dnd.SpellList
|
||||||
|
if err := json.Unmarshal(output.NormalizeOutputs[0].Artifact.Content, &normalized); err != nil {
|
||||||
|
t.Fatalf("decode normalized output: %v", err)
|
||||||
|
}
|
||||||
|
if len(normalized.SpellCasts) != 1 || normalized.SpellCasts[0].Spell != "Mysterious Burst" {
|
||||||
|
t.Fatalf("normalized casts = %#v, want unresolved name preserved", normalized.SpellCasts)
|
||||||
|
}
|
||||||
|
if len(output.Warnings) != 1 || output.Warnings[0].ReasonCode != spellnormalize.ReasonCodeSpellNameUnresolved || output.Warnings[0].Scope != "spell_casts[0]" {
|
||||||
|
t.Fatalf("warnings = %#v, want promoted scoped unresolved-name warning", output.Warnings)
|
||||||
|
}
|
||||||
|
warningsFile := decodeAssembledOutput[struct {
|
||||||
|
Warnings []contracts.Warning `json:"warnings"`
|
||||||
|
}](t, output.OutputFiles, "warnings.json")
|
||||||
|
if !reflect.DeepEqual(warningsFile.Warnings, output.Warnings) {
|
||||||
|
t.Fatalf("warnings file = %#v, run warnings = %#v, want durable unresolved-name warning", warningsFile.Warnings, output.Warnings)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
type assembledSpellPipelineOptions struct {
|
||||||
|
normalizeValidatorOverride bool
|
||||||
|
unknownSpell bool
|
||||||
|
}
|
||||||
|
|
||||||
|
func assembledSpellPipeline(t *testing.T, options assembledSpellPipelineOptions) (pipeline.Registries, pipeline.ResolvedPipeline, *assembledSpellExtractor) {
|
||||||
|
t.Helper()
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
extractor := &assembledSpellExtractor{unknownSpell: options.unknownSpell}
|
||||||
|
if err := pipeline.RegisterExtractor[dnd.SpellList](components.registries.Extractors, pipeline.ModuleSpec{
|
||||||
|
Key: assembledSpellExtractorKey,
|
||||||
|
Stage: pipeline.StageExtract,
|
||||||
|
Requires: []string{"chunks", "source.transcript"},
|
||||||
|
Provides: []string{"dnd.spell_casts"},
|
||||||
|
ArtifactKind: dnd.SpellListKind,
|
||||||
|
}, func() (contracts.Extractor[dnd.SpellList], error) {
|
||||||
|
return extractor, nil
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatalf("register assembled extractor: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
normalize := pipeline.Binding(spellnormalize.Key)
|
||||||
|
if options.normalizeValidatorOverride {
|
||||||
|
normalize.Validators = pipeline.ValidatorOverride{
|
||||||
|
Set: true,
|
||||||
|
Validators: []pipeline.ModuleBinding{pipeline.Binding("generic/always_accept")},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
resolved, err := pipeline.ResolvePipeline(pipeline.PipelineProfile{
|
||||||
|
ID: "assembled-dnd-spells",
|
||||||
|
Input: pipeline.Binding("seriatim"),
|
||||||
|
Chunk: pipeline.ModuleBinding{Module: "generic", Options: map[string]any{"max_units": 1}},
|
||||||
|
Artifacts: map[string]pipeline.ArtifactLaneProfile{
|
||||||
|
"spells": {Extract: pipeline.Binding(assembledSpellExtractorKey), Normalize: normalize},
|
||||||
|
},
|
||||||
|
Output: pipeline.Binding("json"),
|
||||||
|
}, pipeline.ResolveOptions{}, catalogFromRegistries(components.registries))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ResolvePipeline() error = %v, want nil", err)
|
||||||
|
}
|
||||||
|
return components.registries, resolved, extractor
|
||||||
|
}
|
||||||
|
|
||||||
|
type assembledSpellExtractor struct {
|
||||||
|
mu sync.Mutex
|
||||||
|
chunkIndexes []int
|
||||||
|
unknownSpell bool
|
||||||
|
}
|
||||||
|
|
||||||
|
func (e *assembledSpellExtractor) Key() string { return assembledSpellExtractorKey }
|
||||||
|
|
||||||
|
func (*assembledSpellExtractor) ReferenceSlots() []contracts.ReferenceSlot { return nil }
|
||||||
|
|
||||||
|
func (e *assembledSpellExtractor) Extract(ctx context.Context, req contracts.TypedExtractionRequest) (contracts.TypedExtractionResult[dnd.SpellList], error) {
|
||||||
|
if err := ctx.Err(); err != nil {
|
||||||
|
return contracts.TypedExtractionResult[dnd.SpellList]{}, err
|
||||||
|
}
|
||||||
|
if req.Chunk == nil || req.Source == nil {
|
||||||
|
return contracts.TypedExtractionResult[dnd.SpellList]{}, fmt.Errorf("assembled extractor requires source and chunk")
|
||||||
|
}
|
||||||
|
e.mu.Lock()
|
||||||
|
e.chunkIndexes = append(e.chunkIndexes, req.Chunk.Index)
|
||||||
|
e.mu.Unlock()
|
||||||
|
refOne := source.SourceRef{SourceID: req.Source.ID, StartUnitID: 1, EndUnitID: 1}
|
||||||
|
refTwo := source.SourceRef{SourceID: req.Source.ID, StartUnitID: 2, EndUnitID: 2}
|
||||||
|
if e.unknownSpell {
|
||||||
|
if req.Chunk.Index == 0 {
|
||||||
|
return contracts.TypedExtractionResult[dnd.SpellList]{Value: dnd.SpellList{SpellCasts: []dnd.SpellCast{{
|
||||||
|
Caster: "Aria", Spell: "Mysterious Burst", SourceRefs: []source.SourceRef{refOne},
|
||||||
|
}}}}, nil
|
||||||
|
}
|
||||||
|
return contracts.TypedExtractionResult[dnd.SpellList]{Value: dnd.SpellList{SpellCasts: []dnd.SpellCast{}}}, nil
|
||||||
|
}
|
||||||
|
switch req.Chunk.Index {
|
||||||
|
case 0:
|
||||||
|
return contracts.TypedExtractionResult[dnd.SpellList]{Value: dnd.SpellList{SpellCasts: []dnd.SpellCast{{
|
||||||
|
Caster: " Aria \t", Spell: " cure wounds ", SourceRefs: []source.SourceRef{refTwo, refOne},
|
||||||
|
}}}}, nil
|
||||||
|
case 1:
|
||||||
|
return contracts.TypedExtractionResult[dnd.SpellList]{Value: dnd.SpellList{SpellCasts: []dnd.SpellCast{
|
||||||
|
{Caster: "aria", Spell: "Cure Wounds", SourceRefs: []source.SourceRef{refOne, refTwo}},
|
||||||
|
{Caster: "aria", Spell: "Cure Wounds", SourceRefs: []source.SourceRef{refTwo}},
|
||||||
|
}}}, nil
|
||||||
|
default:
|
||||||
|
return contracts.TypedExtractionResult[dnd.SpellList]{}, fmt.Errorf("unexpected assembled chunk index %d", req.Chunk.Index)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (e *assembledSpellExtractor) chunkIndexesSnapshot() []int {
|
||||||
|
e.mu.Lock()
|
||||||
|
defer e.mu.Unlock()
|
||||||
|
return append([]int(nil), e.chunkIndexes...)
|
||||||
|
}
|
||||||
|
|
||||||
|
func decodeAssembledOutput[T any](t *testing.T, files []contracts.OutputFile, name string) T {
|
||||||
|
t.Helper()
|
||||||
|
for _, file := range files {
|
||||||
|
if file.Name != name {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
var value T
|
||||||
|
if err := json.Unmarshal(file.Bytes, &value); err != nil {
|
||||||
|
t.Fatalf("decode %s: %v", name, err)
|
||||||
|
}
|
||||||
|
return value
|
||||||
|
}
|
||||||
|
t.Fatalf("output files = %#v, want %q", files, name)
|
||||||
|
return *new(T)
|
||||||
|
}
|
||||||
371
internal/cli/cache_contract_test.go
Normal file
371
internal/cli/cache_contract_test.go
Normal file
@@ -0,0 +1,371 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/chunkplan"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestRunChunkPlanModePrecedenceAndValidation(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
fileMode string
|
||||||
|
envMode string
|
||||||
|
cliMode string
|
||||||
|
wantStores int
|
||||||
|
}{
|
||||||
|
{name: "default", wantStores: 1},
|
||||||
|
{name: "file", fileMode: "bypass"},
|
||||||
|
{name: "environment", envMode: "bypass"},
|
||||||
|
{name: "cli", envMode: "refresh", cliMode: "bypass"},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
if tt.name == "default" {
|
||||||
|
removeStateTestConfigLine(t, roots.config, " mode: auto\n")
|
||||||
|
} else if tt.fileMode != "" {
|
||||||
|
replaceStateTestConfigLine(t, roots.config, " mode: auto\n", " mode: "+tt.fileMode+"\n")
|
||||||
|
}
|
||||||
|
|
||||||
|
var stores []string
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
opts.LookupEnv = func(name string) (string, bool) {
|
||||||
|
if name == "NOTARIUS_CACHE_CHUNK_PLANS_MODE" && tt.envMode != "" {
|
||||||
|
return tt.envMode, true
|
||||||
|
}
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
opts.ChunkPlanStoreFactory = func(root string) (pipeline.ChunkPlanStore, error) {
|
||||||
|
stores = append(stores, root)
|
||||||
|
return chunkplan.NewFilesystemStore(root)
|
||||||
|
}
|
||||||
|
|
||||||
|
args := []string{"run", "sample", "--config", roots.config, "--input", roots.input}
|
||||||
|
if tt.cliMode != "" {
|
||||||
|
args = append(args, "--chunk_cache", tt.cliMode)
|
||||||
|
}
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
if code := RunWithOptions(args, &stdout, &stderr, opts); code != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
assertStateTestOutput(t, roots.output)
|
||||||
|
if len(stores) != tt.wantStores {
|
||||||
|
t.Fatalf("chunk plan store roots = %v, want %d stores", stores, tt.wantStores)
|
||||||
|
}
|
||||||
|
if tt.wantStores == 1 && stores[0] != roots.plans {
|
||||||
|
t.Fatalf("chunk plan store root = %q, want %q", stores[0], roots.plans)
|
||||||
|
}
|
||||||
|
if tt.wantStores == 0 {
|
||||||
|
assertAbsent(t, roots.plans)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
t.Run("invalid cli syntax is a usage error", func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "invalid"}, &stdout, &stderr, newStateTestHarness().options())
|
||||||
|
if code != 2 || stdout.Len() != 0 || stderr.Len() == 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
assertNoRunState(t, roots)
|
||||||
|
})
|
||||||
|
|
||||||
|
for _, tt := range []struct {
|
||||||
|
name string
|
||||||
|
fileConfig bool
|
||||||
|
}{
|
||||||
|
{name: "invalid environment mode"},
|
||||||
|
{name: "invalid file mode", fileConfig: true},
|
||||||
|
} {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
if tt.fileConfig {
|
||||||
|
replaceStateTestConfigLine(t, roots.config, " mode: auto\n", " mode: invalid\n")
|
||||||
|
} else {
|
||||||
|
opts.LookupEnv = func(name string) (string, bool) {
|
||||||
|
if name == "NOTARIUS_CACHE_CHUNK_PLANS_MODE" {
|
||||||
|
return "invalid", true
|
||||||
|
}
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input}, &stdout, &stderr, opts)
|
||||||
|
if code != 1 || stdout.Len() != 0 || stderr.Len() == 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
assertNoRunState(t, roots)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunChunkPlanRootSelectionAndFailures(t *testing.T) {
|
||||||
|
t.Run("empty configured root uses the per-user cache root", func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
removeStateTestConfigLine(t, roots.config, fmt.Sprintf(" directory: %q\n", roots.plans))
|
||||||
|
userCache := filepath.Join(t.TempDir(), "user-cache")
|
||||||
|
var stores []string
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
opts.UserCacheDir = func() (string, error) { return userCache, nil }
|
||||||
|
opts.ChunkPlanStoreFactory = func(root string) (pipeline.ChunkPlanStore, error) {
|
||||||
|
stores = append(stores, root)
|
||||||
|
return chunkplan.NewFilesystemStore(root)
|
||||||
|
}
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input}, &stdout, &stderr, opts)
|
||||||
|
if code != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
wantRoot := filepath.Join(userCache, "notarius", "chunk-plans")
|
||||||
|
if len(stores) != 1 || stores[0] != wantRoot {
|
||||||
|
t.Fatalf("chunk plan store roots = %v, want [%q]", stores, wantRoot)
|
||||||
|
}
|
||||||
|
assertFile(t, filepath.Join(wantRoot, strings.TrimPrefix(stateTestDigest, "sha256:"), "plan.json"))
|
||||||
|
assertAbsent(t, roots.plans)
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("bypass avoids default cache dependencies", func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
removeStateTestConfigLine(t, roots.config, fmt.Sprintf(" directory: %q\n", roots.plans))
|
||||||
|
userCacheCalls := 0
|
||||||
|
storeCalls := 0
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
opts.UserCacheDir = func() (string, error) {
|
||||||
|
userCacheCalls++
|
||||||
|
return "", errors.New("user cache must not be resolved")
|
||||||
|
}
|
||||||
|
opts.ChunkPlanStoreFactory = func(string) (pipeline.ChunkPlanStore, error) {
|
||||||
|
storeCalls++
|
||||||
|
return nil, errors.New("chunk plan store must not be constructed")
|
||||||
|
}
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass"}, &stdout, &stderr, opts)
|
||||||
|
if code != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
if userCacheCalls != 0 || storeCalls != 0 {
|
||||||
|
t.Fatalf("user cache calls=%d store calls=%d, want none", userCacheCalls, storeCalls)
|
||||||
|
}
|
||||||
|
assertStateTestOutput(t, roots.output)
|
||||||
|
assertAbsent(t, roots.plans)
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("user cache resolution failure has context and no output", func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
removeStateTestConfigLine(t, roots.config, fmt.Sprintf(" directory: %q\n", roots.plans))
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
opts.UserCacheDir = func() (string, error) { return "", errors.New("cache home unavailable") }
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input}, &stdout, &stderr, opts)
|
||||||
|
if code != 1 || !strings.Contains(stderr.String(), "resolve chunk plan root") || !strings.Contains(stderr.String(), "cache home unavailable") {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
assertNoRunState(t, roots)
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("store construction failure has context and no output", func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
opts.ChunkPlanStoreFactory = func(root string) (pipeline.ChunkPlanStore, error) {
|
||||||
|
return nil, fmt.Errorf("store unavailable")
|
||||||
|
}
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input}, &stdout, &stderr, opts)
|
||||||
|
want := fmt.Sprintf("create chunk plan store at %q", roots.plans)
|
||||||
|
if code != 1 || !strings.Contains(stderr.String(), want) || !strings.Contains(stderr.String(), "store unavailable") {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
assertNoRunState(t, roots)
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("checkpoint root resolution failure has context and no output", func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
removeStateTestConfigLine(t, roots.config, fmt.Sprintf(" directory: %q\n", roots.checkpoints))
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
opts.UserCacheDir = func() (string, error) { return "", errors.New("checkpoint cache unavailable") }
|
||||||
|
result := runStateTest(t, roots, opts, false, true, "bypass")
|
||||||
|
if result.code != 1 || !strings.Contains(result.stderr, "resolve checkpoint root") || !strings.Contains(result.stderr, "checkpoint cache unavailable") {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", result.code, result.stdout, result.stderr)
|
||||||
|
}
|
||||||
|
assertNoRunState(t, roots)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunAutoReusesPlanWhenRunInputsChange(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
data, err := os.ReadFile(roots.config)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
configText := replaceRequiredOnce(t, string(data), " chunk: test/chunk\n", ` chunk:
|
||||||
|
module: test/chunk
|
||||||
|
options:
|
||||||
|
strategy: first
|
||||||
|
`)
|
||||||
|
configText = replaceRequiredOnce(t, configText, " output: test/output\n", ` other:
|
||||||
|
extract: test/extract
|
||||||
|
merge: test/merge
|
||||||
|
normalize: test/normalize
|
||||||
|
output: test/output
|
||||||
|
`)
|
||||||
|
if err := os.WriteFile(roots.config, []byte(configText), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
referencePath := filepath.Join(filepath.Dir(roots.input), "reference.txt")
|
||||||
|
if err := os.WriteFile(referencePath, []byte("reference content"), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
var firstStdout, firstStderr bytes.Buffer
|
||||||
|
first := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input}, &firstStdout, &firstStderr, harness.options())
|
||||||
|
if first != 0 {
|
||||||
|
t.Fatalf("first run code=%d stdout=%q stderr=%q", first, firstStdout.String(), firstStderr.String())
|
||||||
|
}
|
||||||
|
configData, err := os.ReadFile(roots.config)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
configText = replaceRequiredOnce(t, string(configData), "strategy: first", "strategy: second")
|
||||||
|
if err := os.WriteFile(roots.config, []byte(configText), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
second := RunWithOptions([]string{
|
||||||
|
"run", "sample", "--config", roots.config, "--input", roots.input,
|
||||||
|
"--only", "items", "--reference", "chunk.cache-reference=" + referencePath,
|
||||||
|
}, &stdout, &stderr, harness.options())
|
||||||
|
if second != 0 {
|
||||||
|
t.Fatalf("second run code=%d stdout=%q stderr=%q", second, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
harness.mu.Lock()
|
||||||
|
chunkCalls := harness.chunkCalls
|
||||||
|
harness.mu.Unlock()
|
||||||
|
if chunkCalls != 1 {
|
||||||
|
t.Fatalf("chunk calls across changed run inputs = %d, want 1", chunkCalls)
|
||||||
|
}
|
||||||
|
assertFile(t, filepath.Join(roots.plans, strings.TrimPrefix(stateTestDigest, "sha256:"), "plan.json"))
|
||||||
|
assertAnyFile(t, roots.output)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunResumeSelectsConfiguredOrPerUserCheckpointRoot(t *testing.T) {
|
||||||
|
for _, configured := range []bool{true, false} {
|
||||||
|
name := "per-user root"
|
||||||
|
if configured {
|
||||||
|
name = "configured root"
|
||||||
|
}
|
||||||
|
t.Run(name, func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
if !configured {
|
||||||
|
removeStateTestConfigLine(t, roots.config, fmt.Sprintf(" directory: %q\n", roots.checkpoints))
|
||||||
|
}
|
||||||
|
userCache := filepath.Join(t.TempDir(), "user-cache")
|
||||||
|
userCacheCalls := 0
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
opts.UserCacheDir = func() (string, error) {
|
||||||
|
userCacheCalls++
|
||||||
|
return userCache, nil
|
||||||
|
}
|
||||||
|
result := runStateTest(t, roots, opts, false, true, "bypass")
|
||||||
|
if result.code != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", result.code, result.stdout, result.stderr)
|
||||||
|
}
|
||||||
|
wantRoot := roots.checkpoints
|
||||||
|
wantCalls := 0
|
||||||
|
if !configured {
|
||||||
|
wantRoot = filepath.Join(userCache, "notarius", "checkpoints")
|
||||||
|
wantCalls = 1
|
||||||
|
}
|
||||||
|
if userCacheCalls != wantCalls {
|
||||||
|
t.Fatalf("user cache calls = %d, want %d", userCacheCalls, wantCalls)
|
||||||
|
}
|
||||||
|
assertAnyFile(t, wantRoot)
|
||||||
|
if !configured {
|
||||||
|
assertAbsent(t, roots.checkpoints)
|
||||||
|
}
|
||||||
|
assertStateTestOutput(t, roots.output)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
t.Run("disabled avoids checkpoint root resolution", func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
replaceStateTestConfigLine(t, roots.config, " enabled: true\n", " enabled: false\n")
|
||||||
|
removeStateTestConfigLine(t, roots.config, fmt.Sprintf(" directory: %q\n", roots.checkpoints))
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
opts.UserCacheDir = func() (string, error) { return "", errors.New("checkpoint cache must not be resolved") }
|
||||||
|
result := runStateTest(t, roots, opts, false, false, "bypass")
|
||||||
|
if result.code != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", result.code, result.stdout, result.stderr)
|
||||||
|
}
|
||||||
|
assertStateTestOutput(t, roots.output)
|
||||||
|
assertAbsent(t, roots.checkpoints)
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("resume requires enabled checkpoint recording", func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
replaceStateTestConfigLine(t, roots.config, " enabled: true\n", " enabled: false\n")
|
||||||
|
result := runStateTest(t, roots, newStateTestHarness().options(), true, true, "bypass")
|
||||||
|
if result.code != 1 || !strings.Contains(result.stderr, "--resume requires cache.checkpoints.enabled: true") {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", result.code, result.stdout, result.stderr)
|
||||||
|
}
|
||||||
|
assertNoRunState(t, roots)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestConfigCommandsDoNotResolveRunState(t *testing.T) {
|
||||||
|
for _, args := range [][]string{
|
||||||
|
{"config", "validate", "--config"},
|
||||||
|
{"pipelines", "list", "--config"},
|
||||||
|
} {
|
||||||
|
name := strings.Join(args[:2], "-")
|
||||||
|
t.Run(name, func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
opts.UserCacheDir = func() (string, error) { return "", errors.New("state root must not be resolved") }
|
||||||
|
opts.ChunkPlanStoreFactory = func(string) (pipeline.ChunkPlanStore, error) {
|
||||||
|
return nil, errors.New("chunk plan store must not be constructed")
|
||||||
|
}
|
||||||
|
command := append([]string(nil), args...)
|
||||||
|
command = append(command, roots.config)
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions(command, &stdout, &stderr, opts)
|
||||||
|
if code != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
assertNoRunState(t, roots)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func replaceStateTestConfigLine(t *testing.T, path, old, new string) {
|
||||||
|
t.Helper()
|
||||||
|
data, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
text := replaceRequiredOnce(t, string(data), old, new)
|
||||||
|
if err := os.WriteFile(path, []byte(text), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func removeStateTestConfigLine(t *testing.T, path, line string) {
|
||||||
|
replaceStateTestConfigLine(t, path, line, "")
|
||||||
|
}
|
||||||
|
|
||||||
|
func assertNoRunState(t *testing.T, roots stateTestRoots) {
|
||||||
|
t.Helper()
|
||||||
|
assertAbsent(t, roots.output)
|
||||||
|
assertAbsent(t, roots.plans)
|
||||||
|
assertAbsent(t, roots.checkpoints)
|
||||||
|
assertAbsent(t, roots.debug)
|
||||||
|
}
|
||||||
@@ -3,50 +3,55 @@ package cli
|
|||||||
import (
|
import (
|
||||||
"context"
|
"context"
|
||||||
"fmt"
|
"fmt"
|
||||||
"strings"
|
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/chunk/generic"
|
dndregister "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/register"
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/extract/dnd/spells"
|
genericregister "gitea.maximumdirect.net/eric/notarius/internal/modules/generic/register"
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/input/seriatim"
|
seriatimregister "gitea.maximumdirect.net/eric/notarius/internal/modules/seriatim/register"
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/merge/appendorder"
|
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/normalize/noop"
|
|
||||||
jsonoutput "gitea.maximumdirect.net/eric/notarius/internal/modules/output/json"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
func productionRegistries() (pipeline.Registries, error) {
|
type productionComponents struct {
|
||||||
|
registries pipeline.Registries
|
||||||
|
assets *llm.AssetRegistry
|
||||||
|
}
|
||||||
|
|
||||||
|
func newProductionComponents() (productionComponents, error) {
|
||||||
registries := pipeline.Registries{
|
registries := pipeline.Registries{
|
||||||
Inputs: pipeline.NewInputAdapterRegistry(),
|
Inputs: pipeline.NewInputAdapterRegistry(),
|
||||||
Chunkers: pipeline.NewChunkerRegistry(),
|
Chunkers: pipeline.NewChunkerRegistry(),
|
||||||
|
ArtifactCodecs: pipeline.NewArtifactCodecRegistry(),
|
||||||
|
ArtifactEvidence: pipeline.NewArtifactEvidenceRegistry(),
|
||||||
Extractors: pipeline.NewExtractorRegistry(),
|
Extractors: pipeline.NewExtractorRegistry(),
|
||||||
Mergers: pipeline.NewMergerRegistry(),
|
Mergers: pipeline.NewMergerRegistry(),
|
||||||
Normalizers: pipeline.NewNormalizerRegistry(),
|
Normalizers: pipeline.NewNormalizerRegistry(),
|
||||||
Validators: pipeline.NewValidatorRegistry(),
|
Validators: pipeline.NewValidatorRegistry(),
|
||||||
|
ValidatorChains: pipeline.NewValidatorChainRegistry(),
|
||||||
Outputs: pipeline.NewOutputEncoderRegistry(),
|
Outputs: pipeline.NewOutputEncoderRegistry(),
|
||||||
}
|
}
|
||||||
if err := seriatim.Register(registries.Inputs); err != nil {
|
assets := llm.NewAssetRegistry()
|
||||||
return pipeline.Registries{}, fmt.Errorf("register seriatim input: %w", err)
|
registrars := []struct {
|
||||||
|
name string
|
||||||
|
register func(pipeline.Registries, *llm.AssetRegistry) error
|
||||||
|
}{
|
||||||
|
{name: "generic", register: genericregister.Register},
|
||||||
|
{name: "seriatim", register: seriatimregister.Register},
|
||||||
|
{name: "dnd", register: dndregister.Register},
|
||||||
}
|
}
|
||||||
if err := generic.Register(registries.Chunkers); err != nil {
|
for _, registrar := range registrars {
|
||||||
return pipeline.Registries{}, fmt.Errorf("register generic chunker: %w", err)
|
if err := registrar.register(registries, assets); err != nil {
|
||||||
|
return productionComponents{}, fmt.Errorf("register %s module family: %w", registrar.name, err)
|
||||||
}
|
}
|
||||||
if err := spells.Register(registries.Extractors); err != nil {
|
|
||||||
return pipeline.Registries{}, fmt.Errorf("register dnd spells extractor: %w", err)
|
|
||||||
}
|
}
|
||||||
if err := appendorder.Register(registries.Mergers); err != nil {
|
return productionComponents{registries: registries, assets: assets}, nil
|
||||||
return pipeline.Registries{}, fmt.Errorf("register appendorder merger: %w", err)
|
|
||||||
}
|
}
|
||||||
if err := noop.Register(registries.Normalizers); err != nil {
|
|
||||||
return pipeline.Registries{}, fmt.Errorf("register noop normalizer: %w", err)
|
func productionRegistries() (pipeline.Registries, error) {
|
||||||
}
|
components, err := newProductionComponents()
|
||||||
if err := jsonoutput.Register(registries.Outputs); err != nil {
|
return components.registries, err
|
||||||
return pipeline.Registries{}, fmt.Errorf("register json output encoder: %w", err)
|
|
||||||
}
|
|
||||||
return registries, nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func productionCatalog() (pipeline.ModuleCatalog, error) {
|
func productionCatalog() (pipeline.ModuleCatalog, error) {
|
||||||
@@ -57,6 +62,11 @@ func productionCatalog() (pipeline.ModuleCatalog, error) {
|
|||||||
return catalogFromRegistries(registries), nil
|
return catalogFromRegistries(registries), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func productionPromptAssets() (*llm.AssetRegistry, error) {
|
||||||
|
components, err := newProductionComponents()
|
||||||
|
return components.assets, err
|
||||||
|
}
|
||||||
|
|
||||||
func effectiveCatalog(opts Options) (pipeline.ModuleCatalog, error) {
|
func effectiveCatalog(opts Options) (pipeline.ModuleCatalog, error) {
|
||||||
if !isEmptyCatalog(opts.Catalog) {
|
if !isEmptyCatalog(opts.Catalog) {
|
||||||
return opts.Catalog, nil
|
return opts.Catalog, nil
|
||||||
@@ -81,10 +91,13 @@ func catalogFromRegistries(registries pipeline.Registries) pipeline.ModuleCatalo
|
|||||||
return pipeline.ModuleCatalog{
|
return pipeline.ModuleCatalog{
|
||||||
Inputs: registries.Inputs,
|
Inputs: registries.Inputs,
|
||||||
Chunkers: registries.Chunkers,
|
Chunkers: registries.Chunkers,
|
||||||
|
ArtifactCodecs: registries.ArtifactCodecs,
|
||||||
|
ArtifactEvidence: registries.ArtifactEvidence,
|
||||||
Extractors: registries.Extractors,
|
Extractors: registries.Extractors,
|
||||||
Mergers: registries.Mergers,
|
Mergers: registries.Mergers,
|
||||||
Normalizers: registries.Normalizers,
|
Normalizers: registries.Normalizers,
|
||||||
Validators: registries.Validators,
|
Validators: registries.Validators,
|
||||||
|
ValidatorChains: registries.ValidatorChains,
|
||||||
Outputs: registries.Outputs,
|
Outputs: registries.Outputs,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -93,10 +106,13 @@ func registriesFromCatalog(catalog pipeline.ModuleCatalog) pipeline.Registries {
|
|||||||
return pipeline.Registries{
|
return pipeline.Registries{
|
||||||
Inputs: catalog.Inputs,
|
Inputs: catalog.Inputs,
|
||||||
Chunkers: catalog.Chunkers,
|
Chunkers: catalog.Chunkers,
|
||||||
|
ArtifactCodecs: catalog.ArtifactCodecs,
|
||||||
|
ArtifactEvidence: catalog.ArtifactEvidence,
|
||||||
Extractors: catalog.Extractors,
|
Extractors: catalog.Extractors,
|
||||||
Mergers: catalog.Mergers,
|
Mergers: catalog.Mergers,
|
||||||
Normalizers: catalog.Normalizers,
|
Normalizers: catalog.Normalizers,
|
||||||
Validators: catalog.Validators,
|
Validators: catalog.Validators,
|
||||||
|
ValidatorChains: catalog.ValidatorChains,
|
||||||
Outputs: catalog.Outputs,
|
Outputs: catalog.Outputs,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -104,20 +120,26 @@ func registriesFromCatalog(catalog pipeline.ModuleCatalog) pipeline.Registries {
|
|||||||
func isEmptyCatalog(catalog pipeline.ModuleCatalog) bool {
|
func isEmptyCatalog(catalog pipeline.ModuleCatalog) bool {
|
||||||
return catalog.Inputs == nil &&
|
return catalog.Inputs == nil &&
|
||||||
catalog.Chunkers == nil &&
|
catalog.Chunkers == nil &&
|
||||||
|
catalog.ArtifactCodecs == nil &&
|
||||||
|
catalog.ArtifactEvidence == nil &&
|
||||||
catalog.Extractors == nil &&
|
catalog.Extractors == nil &&
|
||||||
catalog.Mergers == nil &&
|
catalog.Mergers == nil &&
|
||||||
catalog.Normalizers == nil &&
|
catalog.Normalizers == nil &&
|
||||||
catalog.Validators == nil &&
|
catalog.Validators == nil &&
|
||||||
|
catalog.ValidatorChains == nil &&
|
||||||
catalog.Outputs == nil
|
catalog.Outputs == nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func isEmptyRegistries(registries pipeline.Registries) bool {
|
func isEmptyRegistries(registries pipeline.Registries) bool {
|
||||||
return registries.Inputs == nil &&
|
return registries.Inputs == nil &&
|
||||||
registries.Chunkers == nil &&
|
registries.Chunkers == nil &&
|
||||||
|
registries.ArtifactCodecs == nil &&
|
||||||
|
registries.ArtifactEvidence == nil &&
|
||||||
registries.Extractors == nil &&
|
registries.Extractors == nil &&
|
||||||
registries.Mergers == nil &&
|
registries.Mergers == nil &&
|
||||||
registries.Normalizers == nil &&
|
registries.Normalizers == nil &&
|
||||||
registries.Validators == nil &&
|
registries.Validators == nil &&
|
||||||
|
registries.ValidatorChains == nil &&
|
||||||
registries.Outputs == nil
|
registries.Outputs == nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -125,48 +147,39 @@ func productionLLMClientFactory(ctx context.Context, cfg config.Config, profileI
|
|||||||
if err := ctx.Err(); err != nil {
|
if err := ctx.Err(); err != nil {
|
||||||
return nil, nil, err
|
return nil, nil, err
|
||||||
}
|
}
|
||||||
trimmedID := strings.TrimSpace(profileID)
|
assets, err := productionPromptAssets()
|
||||||
if trimmedID == "" {
|
|
||||||
trimmedID = pipeline.DefaultLLMProfile
|
|
||||||
}
|
|
||||||
|
|
||||||
profile, ok := cfg.LLMProfile(trimmedID)
|
|
||||||
if !ok {
|
|
||||||
return nil, nil, fmt.Errorf("LLM profile %q is not configured", trimmedID)
|
|
||||||
}
|
|
||||||
clientCfg, err := cfg.OpenAICompatibleClientConfig(trimmedID)
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, nil, err
|
return nil, nil, err
|
||||||
}
|
}
|
||||||
client, err := llm.NewOpenAICompatibleClient(clientCfg)
|
return buildProductionLLMClient(ctx, cfg, profileID, assets)
|
||||||
if err != nil {
|
|
||||||
return nil, nil, fmt.Errorf("create LLM client for profile %q: %w", trimmedID, err)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
scheduler, err := llm.NewScheduler(effectiveLLMConcurrency(cfg, profile))
|
func productionLLMClientFactoryWithAssets(assets *llm.AssetRegistry) LLMClientFactory {
|
||||||
if err != nil {
|
return func(ctx context.Context, cfg config.Config, profileID string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||||
return nil, nil, fmt.Errorf("create LLM scheduler for profile %q: %w", trimmedID, err)
|
return buildProductionLLMClient(ctx, cfg, profileID, assets)
|
||||||
}
|
}
|
||||||
provider := strings.TrimSpace(profile.Provider)
|
|
||||||
if provider == "" {
|
|
||||||
provider = "openai-compatible"
|
|
||||||
}
|
|
||||||
metadata := []artifacts.LLMProfileManifest{
|
|
||||||
{
|
|
||||||
ID: trimmedID,
|
|
||||||
Provider: provider,
|
|
||||||
Model: strings.TrimSpace(profile.Model),
|
|
||||||
},
|
|
||||||
}
|
|
||||||
return llm.NewScheduledClient(client, scheduler), metadata, nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func effectiveLLMConcurrency(cfg config.Config, profile config.LLMProfile) int {
|
func buildProductionLLMClient(ctx context.Context, cfg config.Config, profileID string, assets *llm.AssetRegistry) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||||
if profile.MaxConcurrency > 0 {
|
if err := ctx.Err(); err != nil {
|
||||||
return profile.MaxConcurrency
|
return nil, nil, err
|
||||||
}
|
}
|
||||||
if cfg.Concurrency.TotalLLM > 0 {
|
if assets == nil {
|
||||||
return cfg.Concurrency.TotalLLM
|
return nil, nil, fmt.Errorf("production asset registry must not be nil")
|
||||||
}
|
}
|
||||||
return 1
|
recorder := llm.NewLLMProfileRecorder()
|
||||||
|
client, err := llm.NewScriptoriumClient(llm.ScriptoriumClientConfig{
|
||||||
|
ProfileDir: cfg.Scriptorium.ProfileDir,
|
||||||
|
ProfileFile: cfg.Scriptorium.ProfileFile,
|
||||||
|
Assets: assets,
|
||||||
|
Recorder: recorder,
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, fmt.Errorf("create Scriptorium-backed LLM client: %w", err)
|
||||||
|
}
|
||||||
|
scheduler, err := llm.NewScheduler(cfg.Concurrency.TotalLLM)
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, fmt.Errorf("create LLM scheduler: %w", err)
|
||||||
|
}
|
||||||
|
return llm.NewScheduledClient(client, scheduler), nil, nil
|
||||||
}
|
}
|
||||||
|
|||||||
240
internal/cli/command_contract_test.go
Normal file
240
internal/cli/command_contract_test.go
Normal file
@@ -0,0 +1,240 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestCommandHelpSpellingsWriteUsageToStdout(t *testing.T) {
|
||||||
|
tests := [][]string{nil, {"help"}, {"--help"}, {"-h"}}
|
||||||
|
for _, args := range tests {
|
||||||
|
name := "no arguments"
|
||||||
|
if len(args) > 0 {
|
||||||
|
name = args[0]
|
||||||
|
}
|
||||||
|
t.Run(name, func(t *testing.T) {
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions(args, &stdout, &stderr, commandContractOptions(t))
|
||||||
|
if code != 0 || !strings.Contains(stdout.String(), "Usage:") || stderr.Len() != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestCommandSyntaxErrorsUseStderrAndExitTwo(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
args []string
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{name: "unknown command", args: []string{"unknown"}, want: "unknown command"},
|
||||||
|
{name: "missing config subcommand", args: []string{"config"}, want: "config requires a subcommand"},
|
||||||
|
{name: "unknown pipelines subcommand", args: []string{"pipelines", "unknown"}, want: "unknown pipelines subcommand"},
|
||||||
|
{name: "malformed run flag", args: []string{"run", "demo", "--chunk_cache", "invalid"}, want: "not supported"},
|
||||||
|
{name: "unknown flag", args: []string{"config", "validate", "--unknown"}, want: "flag provided but not defined"},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions(tt.args, &stdout, &stderr, commandContractOptions(t))
|
||||||
|
if code != 2 || !strings.Contains(stderr.String(), tt.want) || stdout.Len() != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestConfigDiscoveryPrefersExplicitPathThenEnvironment(t *testing.T) {
|
||||||
|
explicit := writeCommandConfig(t, "explicit", "alpha")
|
||||||
|
environment := writeCommandConfig(t, "environment", "beta")
|
||||||
|
lookup := func(name string) (string, bool) {
|
||||||
|
if name == "NOTARIUS_CONFIG" {
|
||||||
|
return environment, true
|
||||||
|
}
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"pipelines", "list", "--config", explicit}, &stdout, &stderr, commandContractOptionsWithLookup(t, lookup))
|
||||||
|
if code != 0 || stdout.String() != "alpha\nexplicit\n" || stderr.Len() != 0 {
|
||||||
|
t.Fatalf("explicit config: code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
stdout.Reset()
|
||||||
|
stderr.Reset()
|
||||||
|
code = RunWithOptions([]string{"pipelines", "list"}, &stdout, &stderr, commandContractOptionsWithLookup(t, lookup))
|
||||||
|
if code != 0 || stdout.String() != "beta\nenvironment\n" || stderr.Len() != 0 {
|
||||||
|
t.Fatalf("environment config: code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestConfigDiscoveryUsesCompiledDefaultOnlyWhenAvailable(t *testing.T) {
|
||||||
|
info, statErr := os.Stat(defaultConfigPath)
|
||||||
|
if statErr != nil && !os.IsNotExist(statErr) {
|
||||||
|
t.Fatalf("stat compiled default config: %v", statErr)
|
||||||
|
}
|
||||||
|
if statErr == nil && !info.Mode().IsRegular() {
|
||||||
|
t.Skipf("compiled default config has unexpected host state: %s", info.Mode())
|
||||||
|
}
|
||||||
|
|
||||||
|
path, err := discoverConfigPath("", commandContractOptions(t))
|
||||||
|
if statErr == nil {
|
||||||
|
if err != nil || path != defaultConfigPath {
|
||||||
|
t.Fatalf("discoverConfigPath() = %q, %v; want compiled default", path, err)
|
||||||
|
}
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "config file not found") {
|
||||||
|
t.Fatalf("discoverConfigPath() error = %v, want documented not-found context", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestConfigLoadingFailuresReturnOneWithPathContext(t *testing.T) {
|
||||||
|
missing := filepath.Join(t.TempDir(), "missing.yml")
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"config", "validate", "--config", missing}, &stdout, &stderr, commandContractOptions(t))
|
||||||
|
if code != 1 || !strings.Contains(stderr.String(), missing) || stdout.Len() != 0 {
|
||||||
|
t.Fatalf("missing config: code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
malformed := filepath.Join(t.TempDir(), "malformed.yml")
|
||||||
|
if err := os.WriteFile(malformed, []byte("version: [\n"), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
stdout.Reset()
|
||||||
|
stderr.Reset()
|
||||||
|
code = RunWithOptions([]string{"config", "validate", "--config", malformed}, &stdout, &stderr, commandContractOptions(t))
|
||||||
|
if code != 1 || !strings.Contains(stderr.String(), malformed) || !strings.Contains(stderr.String(), "parse config file") || stdout.Len() != 0 {
|
||||||
|
t.Fatalf("malformed config: code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestConfigValidateResolvesPipelineAndChecksSelection(t *testing.T) {
|
||||||
|
configPath := writeResolvableCommandConfig(t)
|
||||||
|
options := commandContractOptions(t)
|
||||||
|
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"config", "validate", "--config", configPath, "--pipeline", "demo", "--only", "spells"}, &stdout, &stderr, options)
|
||||||
|
if code != 0 || !strings.Contains(stdout.String(), "valid for pipeline \"demo\"") || stderr.Len() != 0 {
|
||||||
|
t.Fatalf("valid resolution: code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
stdout.Reset()
|
||||||
|
stderr.Reset()
|
||||||
|
code = RunWithOptions([]string{"config", "validate", "--config", configPath, "--pipeline", "missing"}, &stdout, &stderr, options)
|
||||||
|
if code != 1 || !strings.Contains(stderr.String(), "pipeline \"missing\"") {
|
||||||
|
t.Fatalf("unknown pipeline: code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
stdout.Reset()
|
||||||
|
stderr.Reset()
|
||||||
|
code = RunWithOptions([]string{"config", "validate", "--config", configPath, "--pipeline", "demo", "--only", "missing"}, &stdout, &stderr, options)
|
||||||
|
if code != 1 || !strings.Contains(stderr.String(), "lane \"missing\"") {
|
||||||
|
t.Fatalf("unknown lane: code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
stdout.Reset()
|
||||||
|
stderr.Reset()
|
||||||
|
code = RunWithOptions([]string{"config", "validate", "--config", configPath, "--only", "spells"}, &stdout, &stderr, options)
|
||||||
|
if code != 2 || !strings.Contains(stderr.String(), "--only requires --pipeline") {
|
||||||
|
t.Fatalf("missing pipeline for only: code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
stdout.Reset()
|
||||||
|
stderr.Reset()
|
||||||
|
code = RunWithOptions([]string{"config", "validate", "--config", configPath, "--pipeline", "demo", "--only", "spells,,other"}, &stdout, &stderr, options)
|
||||||
|
if code != 2 || !strings.Contains(stderr.String(), "--only must contain") {
|
||||||
|
t.Fatalf("malformed only: code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestPipelinesListSortsNormalizedIDsInTextAndJSON(t *testing.T) {
|
||||||
|
configPath := writeCommandConfig(t, " zeta ", "alpha")
|
||||||
|
options := commandContractOptions(t)
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"pipelines", "list", "--config", configPath}, &stdout, &stderr, options)
|
||||||
|
if code != 0 || stdout.String() != "alpha\nzeta\n" || stderr.Len() != 0 {
|
||||||
|
t.Fatalf("text list: code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
stdout.Reset()
|
||||||
|
stderr.Reset()
|
||||||
|
code = RunWithOptions([]string{"pipelines", "list", "--config", configPath, "--json"}, &stdout, &stderr, options)
|
||||||
|
var payload struct {
|
||||||
|
Pipelines []string `json:"pipelines"`
|
||||||
|
}
|
||||||
|
if err := json.Unmarshal(stdout.Bytes(), &payload); err != nil {
|
||||||
|
t.Fatalf("JSON list = %q: %v", stdout.String(), err)
|
||||||
|
}
|
||||||
|
if code != 0 || len(payload.Pipelines) != 2 || payload.Pipelines[0] != "alpha" || payload.Pipelines[1] != "zeta" || stderr.Len() != 0 {
|
||||||
|
t.Fatalf("JSON list: code=%d payload=%#v stderr=%q", code, payload, stderr.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRemovedStructuralFlagsAndRuntimeFailuresKeepExitClasses(t *testing.T) {
|
||||||
|
configPath := writeResolvableCommandConfig(t)
|
||||||
|
options := commandContractOptions(t)
|
||||||
|
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"run", "demo", "--input", "missing-input", "--config", configPath, "--diagnostics-dir", t.TempDir()}, &stdout, &stderr, options)
|
||||||
|
if code != 2 || !strings.Contains(stderr.String(), "flag provided but not defined") {
|
||||||
|
t.Fatalf("removed flag: code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
stdout.Reset()
|
||||||
|
stderr.Reset()
|
||||||
|
code = RunWithOptions([]string{"run", "missing", "--input", "missing-input", "--config", configPath, "--chunk_cache", "bypass"}, &stdout, &stderr, options)
|
||||||
|
if code != 1 || !strings.Contains(stderr.String(), "pipeline \"missing\"") || stdout.Len() != 0 {
|
||||||
|
t.Fatalf("valid-runtime failure: code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func commandContractOptions(t *testing.T) Options {
|
||||||
|
return commandContractOptionsWithLookup(t, emptyLookup)
|
||||||
|
}
|
||||||
|
|
||||||
|
func commandContractOptionsWithLookup(t *testing.T, lookup func(string) (string, bool)) Options {
|
||||||
|
t.Helper()
|
||||||
|
components, err := newProductionComponents()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return Options{
|
||||||
|
Catalog: catalogFromRegistries(components.registries),
|
||||||
|
Registries: components.registries,
|
||||||
|
LookupEnv: lookup,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func writeCommandConfig(t *testing.T, firstID, secondID string) string {
|
||||||
|
t.Helper()
|
||||||
|
content := fmt.Sprintf("version: 3\npipelines:\n %q:\n input: seriatim\n %q:\n input: seriatim\n", firstID, secondID)
|
||||||
|
return writeCommandConfigContent(t, content)
|
||||||
|
}
|
||||||
|
|
||||||
|
func writeResolvableCommandConfig(t *testing.T) string {
|
||||||
|
t.Helper()
|
||||||
|
return writeCommandConfigContent(t, `version: 3
|
||||||
|
pipelines:
|
||||||
|
demo:
|
||||||
|
input: seriatim
|
||||||
|
artifacts:
|
||||||
|
spells:
|
||||||
|
extract: dnd/spells
|
||||||
|
`)
|
||||||
|
}
|
||||||
|
|
||||||
|
func writeCommandConfigContent(t *testing.T, content string) string {
|
||||||
|
t.Helper()
|
||||||
|
path := filepath.Join(t.TempDir(), "config.yml")
|
||||||
|
if err := os.WriteFile(path, []byte(content), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return path
|
||||||
|
}
|
||||||
16
internal/cli/contract_test_helpers_test.go
Normal file
16
internal/cli/contract_test_helpers_test.go
Normal file
@@ -0,0 +1,16 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func replaceRequiredOnce(t *testing.T, input, old, replacement string) string {
|
||||||
|
t.Helper()
|
||||||
|
if count := strings.Count(input, old); count != 1 {
|
||||||
|
t.Fatalf("replacement marker %q occurs %d times, want exactly once", old, count)
|
||||||
|
}
|
||||||
|
return strings.Replace(input, old, replacement, 1)
|
||||||
|
}
|
||||||
|
|
||||||
|
func emptyLookup(string) (string, bool) { return "", false }
|
||||||
208
internal/cli/dnd_combat_contract_test.go
Normal file
208
internal/cli/dnd_combat_contract_test.go
Normal file
@@ -0,0 +1,208 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"reflect"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||||
|
combatextract "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/combatturns"
|
||||||
|
combatnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/combatturns"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestProductionCombatConfigurationResolvesTypedLane(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
cfg := productionCombatContractConfig()
|
||||||
|
effective, err := cfg.Resolve(config.ResolveInput{PipelineID: "dnd-combat", Catalog: catalogFromRegistries(components.registries)})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Resolve() error = %v, want nil", err)
|
||||||
|
}
|
||||||
|
if effective.ResolvedPipeline.Chunk.Module != pipeline.DefaultChunkModule {
|
||||||
|
t.Fatalf("chunk module = %q, want %q", effective.ResolvedPipeline.Chunk.Module, pipeline.DefaultChunkModule)
|
||||||
|
}
|
||||||
|
if len(effective.ResolvedPipeline.Steps[0].ArtifactLanes) != 1 {
|
||||||
|
t.Fatalf("artifact lanes = %#v, want one combat lane", effective.ResolvedPipeline.Steps[0].ArtifactLanes)
|
||||||
|
}
|
||||||
|
lane := effective.ResolvedPipeline.Steps[0].ArtifactLanes[0]
|
||||||
|
if lane.ID != "combat" || lane.ArtifactKind != dnd.CombatTurnListKind || lane.Extract.Module != combatextract.Key || lane.Extract.Retries != 2 || lane.Merge.Module != pipeline.DefaultMergeModule || lane.Normalize.Module != combatnormalize.Key {
|
||||||
|
t.Fatalf("resolved combat lane = %#v, want typed production composition", lane)
|
||||||
|
}
|
||||||
|
|
||||||
|
catalog := catalogFromRegistries(components.registries)
|
||||||
|
extractSpec, ok := catalog.Extractors.Spec(combatextract.Key)
|
||||||
|
if !ok || !reflect.DeepEqual(extractSpec.Requires, []string{"chunks", "source.transcript"}) || !reflect.DeepEqual(extractSpec.Provides, []string{"dnd.combat_turns"}) {
|
||||||
|
t.Fatalf("combat extractor spec = %#v, want source and artifact capabilities", extractSpec)
|
||||||
|
}
|
||||||
|
normalizeSpec, ok := catalog.Normalizers.SpecForArtifact(combatnormalize.Key, dnd.CombatTurnListKind)
|
||||||
|
if !ok || !reflect.DeepEqual(normalizeSpec.Requires, []string{"merged"}) || !reflect.DeepEqual(normalizeSpec.Provides, []string{"normalized"}) {
|
||||||
|
t.Fatalf("combat normalizer spec = %#v, want merged/normalized capabilities", normalizeSpec)
|
||||||
|
}
|
||||||
|
mergeSpec, ok := catalog.Mergers.SpecForArtifact(pipeline.DefaultMergeModule, dnd.CombatTurnListKind)
|
||||||
|
if !ok || !reflect.DeepEqual(mergeSpec.Provides, []string{"merged"}) {
|
||||||
|
t.Fatalf("combat merger spec = %#v, want merged capability", mergeSpec)
|
||||||
|
}
|
||||||
|
codecSpec, ok := catalog.ArtifactCodecs.Spec(dnd.CombatTurnListKind)
|
||||||
|
if !ok || codecSpec.Schema.ID != "notarius.dnd.combat_turns" || codecSpec.Schema.Version != "v1" {
|
||||||
|
t.Fatalf("combat codec spec = %#v, want compatible durable schema", codecSpec)
|
||||||
|
}
|
||||||
|
if !hasReferenceSlot(extractSpec.ReferenceSlots, "npcs") || !hasReferenceSlot(extractSpec.ReferenceSlots, "scene_descriptions") || !hasReferenceSlot(normalizeSpec.ReferenceSlots, "npcs") {
|
||||||
|
t.Fatalf("combat reference slots = %#v / %#v, want extraction scene and NPC slots plus normalization NPC slot", extractSpec.ReferenceSlots, normalizeSpec.ReferenceSlots)
|
||||||
|
}
|
||||||
|
sceneSlot := referenceSlot(extractSpec.ReferenceSlots, "scene_descriptions")
|
||||||
|
if !sceneSlot.Required || !reflect.DeepEqual(sceneSlot.AcceptedMediaTypes, []string{"application/json"}) || !reflect.DeepEqual(sceneSlot.AcceptedArtifactKinds, []contracts.ArtifactKind{dnd.SceneDescriptionListKind}) || sceneSlot.MaxBytes != 1048576 {
|
||||||
|
t.Fatalf("scene description slot = %#v, want required approved scene artifact", sceneSlot)
|
||||||
|
}
|
||||||
|
|
||||||
|
wantExtractChain := []pipeline.ModuleBinding{
|
||||||
|
pipeline.Binding("generic/valid_json"),
|
||||||
|
pipeline.Binding("extract/dnd/combat-turns/shape"),
|
||||||
|
pipeline.Binding("extract/dnd/combat-turns/source_refs"),
|
||||||
|
pipeline.Binding("generic/valid_json_schema"),
|
||||||
|
pipeline.Binding("extract/dnd/combat-turns/source_relatedness"),
|
||||||
|
}
|
||||||
|
wantNormalizeChain := []pipeline.ModuleBinding{
|
||||||
|
pipeline.Binding("generic/valid_json"),
|
||||||
|
pipeline.Binding("extract/dnd/combat-turns/shape"),
|
||||||
|
pipeline.Binding("normalize/dnd/combat-turns/invariants"),
|
||||||
|
pipeline.Binding("extract/dnd/combat-turns/source_refs"),
|
||||||
|
pipeline.Binding("generic/valid_json_schema"),
|
||||||
|
pipeline.Binding("extract/dnd/combat-turns/source_relatedness"),
|
||||||
|
}
|
||||||
|
if got := validatorChain(effective.ResolvedPipeline, pipeline.StageExtract, combatextract.Key); !reflect.DeepEqual(got, wantExtractChain) {
|
||||||
|
t.Fatalf("combat extract chain = %#v, want %#v", got, wantExtractChain)
|
||||||
|
}
|
||||||
|
if got := validatorChain(effective.ResolvedPipeline, pipeline.StageNormalize, combatnormalize.Key); !reflect.DeepEqual(got, wantNormalizeChain) {
|
||||||
|
t.Fatalf("combat normalize chain = %#v, want %#v", got, wantNormalizeChain)
|
||||||
|
}
|
||||||
|
if got := validatorChain(effective.ResolvedPipeline, pipeline.StageMerge, pipeline.DefaultMergeModule); len(got) != 0 {
|
||||||
|
t.Fatalf("combat merge chain = %#v, want empty", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
bound, err := cfg.Resolve(config.ResolveInput{
|
||||||
|
PipelineID: "dnd-combat",
|
||||||
|
Catalog: catalog,
|
||||||
|
ReferenceOverrides: []pipeline.ReferenceBinding{
|
||||||
|
{Stage: pipeline.StageExtract, LaneID: "combat", SlotName: "npcs", Source: "npc-run/lanes/npcs.json", BindingSource: contracts.ReferenceBindingSourceCLI},
|
||||||
|
{Stage: pipeline.StageNormalize, LaneID: "combat", SlotName: "npcs", Source: "npc-run/lanes/npcs.json", BindingSource: contracts.ReferenceBindingSourceCLI},
|
||||||
|
},
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Resolve(bound references) error = %v, want nil", err)
|
||||||
|
}
|
||||||
|
boundLane := bound.ResolvedPipeline.Steps[0].ArtifactLanes[0]
|
||||||
|
if len(boundLane.ExtractReferences.Bindings) != 2 || len(boundLane.NormalizeReferences.Bindings) != 1 || !hasReferenceBinding(boundLane.ExtractReferences.Bindings, "npcs") || !hasReferenceBinding(boundLane.ExtractReferences.Bindings, "scene_descriptions") || !hasReferenceBinding(boundLane.NormalizeReferences.Bindings, "npcs") {
|
||||||
|
t.Fatalf("bound combat references = %#v / %#v, want extraction scene and NPC bindings plus normalization NPC binding", boundLane.ExtractReferences, boundLane.NormalizeReferences)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestProductionCombatConfigurationRequiresSceneDescriptions(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
cfg := productionCombatContractConfig()
|
||||||
|
profile := cfg.Pipelines["dnd-combat"]
|
||||||
|
profile.References = nil
|
||||||
|
cfg.Pipelines["dnd-combat"] = profile
|
||||||
|
if _, err := cfg.Resolve(config.ResolveInput{PipelineID: "dnd-combat", Catalog: catalogFromRegistries(components.registries)}); err == nil || !strings.Contains(err.Error(), "scene_descriptions") || !strings.Contains(err.Error(), "required") {
|
||||||
|
t.Fatalf("Resolve() error = %v, want required scene reference failure", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestProductionCombatConfigurationRejectsLooseOptionsAndLaneValidators(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
resolve := func(mutate func(*pipeline.PipelineProfile)) error {
|
||||||
|
cfg := productionCombatContractConfig()
|
||||||
|
profile := cfg.Pipelines["dnd-combat"]
|
||||||
|
mutate(&profile)
|
||||||
|
cfg.Pipelines["dnd-combat"] = profile
|
||||||
|
_, err := cfg.Resolve(config.ResolveInput{PipelineID: "dnd-combat", Catalog: catalogFromRegistries(components.registries)})
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if err := resolve(func(profile *pipeline.PipelineProfile) {
|
||||||
|
lane := profile.Artifacts["combat"]
|
||||||
|
lane.Extract.Options = map[string]any{"unexpected": true}
|
||||||
|
profile.Artifacts["combat"] = lane
|
||||||
|
}); err == nil || !strings.Contains(err.Error(), "unknown option") {
|
||||||
|
t.Fatalf("unknown extractor option error = %v, want strict option rejection", err)
|
||||||
|
}
|
||||||
|
if err := resolve(func(profile *pipeline.PipelineProfile) {
|
||||||
|
lane := profile.Artifacts["combat"]
|
||||||
|
lane.Normalize.Options = map[string]any{"unexpected": true}
|
||||||
|
profile.Artifacts["combat"] = lane
|
||||||
|
}); err == nil || !strings.Contains(err.Error(), "unknown option") {
|
||||||
|
t.Fatalf("unknown normalizer option error = %v, want strict option rejection", err)
|
||||||
|
}
|
||||||
|
if err := resolve(func(profile *pipeline.PipelineProfile) {
|
||||||
|
lane := profile.Artifacts["combat"]
|
||||||
|
lane.Validators = []pipeline.ModuleBinding{pipeline.Binding("generic/always_accept")}
|
||||||
|
profile.Artifacts["combat"] = lane
|
||||||
|
}); err == nil || !strings.Contains(err.Error(), "artifact lane level") {
|
||||||
|
t.Fatalf("lane-level validator error = %v, want invalid placement rejection", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestProductionCombatConfigurationResolvesTypedUnconditionalValidators(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
cfg := productionCombatContractConfig()
|
||||||
|
profile := cfg.Pipelines["dnd-combat"]
|
||||||
|
lane := profile.Artifacts["combat"]
|
||||||
|
lane.Extract.Validators = pipeline.ValidatorOverride{Set: true, Validators: []pipeline.ModuleBinding{pipeline.Binding("generic/always_accept")}}
|
||||||
|
lane.Normalize.Validators = pipeline.ValidatorOverride{Set: true, Validators: []pipeline.ModuleBinding{pipeline.Binding("generic/always_reject")}}
|
||||||
|
profile.Artifacts["combat"] = lane
|
||||||
|
cfg.Pipelines["dnd-combat"] = profile
|
||||||
|
effective, err := cfg.Resolve(config.ResolveInput{PipelineID: "dnd-combat", Catalog: catalogFromRegistries(components.registries)})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Resolve() error = %v, want typed unconditional validators to resolve", err)
|
||||||
|
}
|
||||||
|
if got := validatorChain(effective.ResolvedPipeline, pipeline.StageExtract, combatextract.Key); !reflect.DeepEqual(got, []pipeline.ModuleBinding{pipeline.Binding("generic/always_accept")}) {
|
||||||
|
t.Fatalf("extract override chain = %#v, want typed always-accept", got)
|
||||||
|
}
|
||||||
|
if got := validatorChain(effective.ResolvedPipeline, pipeline.StageNormalize, combatnormalize.Key); !reflect.DeepEqual(got, []pipeline.ModuleBinding{pipeline.Binding("generic/always_reject")}) {
|
||||||
|
t.Fatalf("normalize override chain = %#v, want typed always-reject", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func productionCombatContractConfig() config.Config {
|
||||||
|
cfg := config.Default()
|
||||||
|
cfg.Pipelines["dnd-combat"] = pipeline.PipelineProfile{
|
||||||
|
ID: "dnd-combat",
|
||||||
|
Input: pipeline.Binding("seriatim"),
|
||||||
|
Chunk: pipeline.Binding(pipeline.DefaultChunkModule),
|
||||||
|
References: map[string]pipeline.ReferenceSource{"scene_descriptions": pipeline.ExternalReference("scenes.json")},
|
||||||
|
Artifacts: map[string]pipeline.ArtifactLaneProfile{
|
||||||
|
"combat": {
|
||||||
|
Extract: pipeline.ModuleBinding{Module: combatextract.Key, Retries: 2},
|
||||||
|
Normalize: pipeline.Binding(combatnormalize.Key),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
return cfg
|
||||||
|
}
|
||||||
|
|
||||||
|
func hasReferenceSlot(slots []contracts.ReferenceSlot, name string) bool {
|
||||||
|
for _, slot := range slots {
|
||||||
|
if slot.Name == name {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
func hasReferenceBinding(bindings []pipeline.ReferenceBinding, name string) bool {
|
||||||
|
for _, binding := range bindings {
|
||||||
|
if binding.SlotName == name {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
func referenceSlot(slots []contracts.ReferenceSlot, name string) contracts.ReferenceSlot {
|
||||||
|
for _, slot := range slots {
|
||||||
|
if slot.Name == name {
|
||||||
|
return slot
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return contracts.ReferenceSlot{}
|
||||||
|
}
|
||||||
135
internal/cli/dnd_interactions_contract_test.go
Normal file
135
internal/cli/dnd_interactions_contract_test.go
Normal file
@@ -0,0 +1,135 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||||
|
interactioncodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/npcinteractions"
|
||||||
|
interactionextract "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/npcinteractions"
|
||||||
|
npcextract "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/npcs"
|
||||||
|
interactionnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/npcinteractions"
|
||||||
|
npcnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/npcs"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestProductionNPCInteractionPipelineResolvesAndPrepares(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
resolved, err := pipeline.ResolvePipeline(npcInteractionProfile(pipeline.GeneratedReference("npcs", "npcs")), pipeline.ResolveOptions{}, catalogFromRegistries(components.registries))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ResolvePipeline() error = %v", err)
|
||||||
|
}
|
||||||
|
if len(resolved.Steps) != 2 || len(resolved.Steps[1].ArtifactLanes) != 1 {
|
||||||
|
t.Fatalf("resolved pipeline = %#v", resolved)
|
||||||
|
}
|
||||||
|
lane := resolved.Steps[1].ArtifactLanes[0]
|
||||||
|
if lane.ArtifactKind != dnd.NPCInteractionListKind || lane.Extract.Module != interactionextract.Key || lane.Normalize.Module != interactionnormalize.Key {
|
||||||
|
t.Fatalf("interaction lane = %#v", lane)
|
||||||
|
}
|
||||||
|
for _, bindings := range [][]pipeline.ReferenceBinding{lane.ExtractReferences.Bindings, lane.NormalizeReferences.Bindings} {
|
||||||
|
if len(bindings) != 1 || bindings[0].SlotName != "npcs" || bindings[0].Artifact == nil || bindings[0].Artifact.Step != "npcs" || bindings[0].Artifact.Lane != "npcs" {
|
||||||
|
t.Fatalf("generated bindings = %#v", bindings)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if _, err := pipeline.Prepare(resolved, components.registries, pipeline.ModuleDependencies{LLM: &productionFakeLLMClient{}}); err != nil {
|
||||||
|
t.Fatalf("Prepare() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
catalog := catalogFromRegistries(components.registries)
|
||||||
|
codecSpec, ok := catalog.ArtifactCodecs.Spec(dnd.NPCInteractionListKind)
|
||||||
|
if !ok || codecSpec.Schema.ID != interactioncodec.SchemaID || codecSpec.Schema.Version != interactioncodec.SchemaVersion {
|
||||||
|
t.Fatalf("NPC interaction codec spec = %#v", codecSpec)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestProductionNPCInteractionReferencesRequireEarlierCompatibleProducer(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
catalog := catalogFromRegistries(components.registries)
|
||||||
|
laterProfile := npcInteractionProfile(pipeline.GeneratedReference("npcs", "npcs"))
|
||||||
|
laterProfile.Steps[0].ID = "seed"
|
||||||
|
laterProfile.Steps[0].Artifacts["seed"] = laterProfile.Steps[0].Artifacts["npcs"]
|
||||||
|
delete(laterProfile.Steps[0].Artifacts, "npcs")
|
||||||
|
laterProfile.Steps = append(laterProfile.Steps, pipeline.PipelineStepProfile{ID: "future", Artifacts: map[string]pipeline.ArtifactLaneProfile{
|
||||||
|
"npcs": {Extract: pipeline.Binding(npcextract.Key), Normalize: pipeline.Binding(npcnormalize.Key)},
|
||||||
|
}})
|
||||||
|
laterProfile.Steps[1].References["npcs"] = pipeline.GeneratedReference("future", "npcs")
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
profile pipeline.PipelineProfile
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{name: "missing", profile: npcInteractionProfile(pipeline.ReferenceSource{}), want: "source must not be empty"},
|
||||||
|
{name: "same step", profile: npcInteractionProfile(pipeline.GeneratedReference("interactions", "interactions")), want: "earlier step"},
|
||||||
|
{name: "later step", profile: laterProfile, want: "earlier step"},
|
||||||
|
{name: "wrong artifact kind", profile: npcInteractionProfile(pipeline.GeneratedReference("npcs", "npcs")), want: "does not accept artifact kind"},
|
||||||
|
}
|
||||||
|
tests[3].profile.Steps[0].Artifacts["npcs"] = pipeline.ArtifactLaneProfile{Extract: pipeline.Binding("dnd/spells")}
|
||||||
|
for _, test := range tests {
|
||||||
|
t.Run(test.name, func(t *testing.T) {
|
||||||
|
_, err := pipeline.ResolvePipeline(test.profile, pipeline.ResolveOptions{}, catalog)
|
||||||
|
if err == nil || !strings.Contains(err.Error(), test.want) {
|
||||||
|
t.Fatalf("ResolvePipeline() error = %v, want %q", err, test.want)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestProductionNPCInteractionReferencesRejectIncompatibleExternalRegistries(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
catalog := catalogFromRegistries(components.registries)
|
||||||
|
root := t.TempDir()
|
||||||
|
for _, test := range []struct {
|
||||||
|
name string
|
||||||
|
file string
|
||||||
|
content string
|
||||||
|
prepare bool
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{name: "media type", file: "registry.txt", content: "not JSON", want: "media type"},
|
||||||
|
{name: "artifact schema", file: "registry.json", content: `{"npcs":[{"name":"missing required fields"}]}`, prepare: true, want: "NPC registry"},
|
||||||
|
} {
|
||||||
|
t.Run(test.name, func(t *testing.T) {
|
||||||
|
path := filepath.Join(root, test.file)
|
||||||
|
if err := os.WriteFile(path, []byte(test.content), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
resolved, err := pipeline.ResolvePipeline(npcInteractionProfile(pipeline.ExternalReference(path)), pipeline.ResolveOptions{}, catalog)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ResolvePipeline() error = %v", err)
|
||||||
|
}
|
||||||
|
materialized, _, err := pipeline.MaterializeReferences(resolved, catalog, pipeline.ReferenceMaterializationOptions{})
|
||||||
|
if !test.prepare {
|
||||||
|
if err == nil || !strings.Contains(err.Error(), test.want) {
|
||||||
|
t.Fatalf("MaterializeReferences() error = %v, want %q", err, test.want)
|
||||||
|
}
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("MaterializeReferences() error = %v", err)
|
||||||
|
}
|
||||||
|
if _, err := pipeline.Prepare(materialized, components.registries, pipeline.ModuleDependencies{LLM: &productionFakeLLMClient{}}); err == nil || !strings.Contains(err.Error(), test.want) {
|
||||||
|
t.Fatalf("Prepare() error = %v, want %q", err, test.want)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func npcInteractionProfile(reference pipeline.ReferenceSource) pipeline.PipelineProfile {
|
||||||
|
profile := pipeline.PipelineProfile{
|
||||||
|
ID: "dnd-npc-interactions",
|
||||||
|
Input: pipeline.Binding("seriatim"),
|
||||||
|
Chunk: pipeline.ModuleBinding{Module: "generic", Options: map[string]any{"max_units": 1}},
|
||||||
|
Output: pipeline.Binding("json"),
|
||||||
|
Steps: []pipeline.PipelineStepProfile{
|
||||||
|
{ID: "npcs", Artifacts: map[string]pipeline.ArtifactLaneProfile{
|
||||||
|
"npcs": {Extract: pipeline.Binding(npcextract.Key), Normalize: pipeline.Binding(npcnormalize.Key)},
|
||||||
|
}},
|
||||||
|
{ID: "interactions", References: map[string]pipeline.ReferenceSource{"npcs": reference}, Artifacts: map[string]pipeline.ArtifactLaneProfile{
|
||||||
|
"interactions": {Extract: pipeline.Binding(interactionextract.Key), Normalize: pipeline.Binding(interactionnormalize.Key)},
|
||||||
|
}},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
return profile
|
||||||
|
}
|
||||||
151
internal/cli/dnd_npc_contract_test.go
Normal file
151
internal/cli/dnd_npc_contract_test.go
Normal file
@@ -0,0 +1,151 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"reflect"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||||
|
npccodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/npcs"
|
||||||
|
npcextract "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/npcs"
|
||||||
|
npcnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/npcs"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestProductionNPCConfigurationResolvesTypedLane(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
catalog := catalogFromRegistries(components.registries)
|
||||||
|
cfg := productionNPCContractConfig()
|
||||||
|
effective, err := cfg.Resolve(config.ResolveInput{PipelineID: "dnd-session", Catalog: catalog})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Resolve() error = %v, want nil", err)
|
||||||
|
}
|
||||||
|
if effective.ResolvedPipeline.Chunk.Module != pipeline.DefaultChunkModule {
|
||||||
|
t.Fatalf("chunk module = %q, want %q", effective.ResolvedPipeline.Chunk.Module, pipeline.DefaultChunkModule)
|
||||||
|
}
|
||||||
|
if len(effective.ResolvedPipeline.Steps[0].ArtifactLanes) != 1 {
|
||||||
|
t.Fatalf("artifact lanes = %#v, want one NPC lane", effective.ResolvedPipeline.Steps[0].ArtifactLanes)
|
||||||
|
}
|
||||||
|
lane := effective.ResolvedPipeline.Steps[0].ArtifactLanes[0]
|
||||||
|
if lane.ID != "npcs" || lane.ArtifactKind != dnd.NPCListKind || lane.Extract.Module != npcextract.Key || lane.Extract.Retries != 2 || lane.Merge.Module != pipeline.DefaultMergeModule || lane.Normalize.Module != npcnormalize.Key {
|
||||||
|
t.Fatalf("resolved NPC lane = %#v, want typed production composition", lane)
|
||||||
|
}
|
||||||
|
if len(lane.ExtractReferences.Bindings) != 0 || len(lane.NormalizeReferences.Bindings) != 0 {
|
||||||
|
t.Fatalf("unbound NPC references = %#v / %#v, want none", lane.ExtractReferences, lane.NormalizeReferences)
|
||||||
|
}
|
||||||
|
|
||||||
|
extractSpec, ok := catalog.Extractors.Spec(npcextract.Key)
|
||||||
|
if !ok || !reflect.DeepEqual(extractSpec.Requires, []string{"chunks", "source.transcript"}) || !reflect.DeepEqual(extractSpec.Provides, []string{"dnd.npcs"}) {
|
||||||
|
t.Fatalf("NPC extractor spec = %#v, want source and artifact capabilities", extractSpec)
|
||||||
|
}
|
||||||
|
mergeSpec, ok := catalog.Mergers.SpecForArtifact(pipeline.DefaultMergeModule, dnd.NPCListKind)
|
||||||
|
if !ok || !reflect.DeepEqual(mergeSpec.Provides, []string{"merged"}) {
|
||||||
|
t.Fatalf("NPC merger spec = %#v, want merged capability", mergeSpec)
|
||||||
|
}
|
||||||
|
normalizeSpec, ok := catalog.Normalizers.SpecForArtifact(npcnormalize.Key, dnd.NPCListKind)
|
||||||
|
if !ok || !reflect.DeepEqual(normalizeSpec.Requires, []string{"merged"}) || !reflect.DeepEqual(normalizeSpec.Provides, []string{"normalized"}) {
|
||||||
|
t.Fatalf("NPC normalizer spec = %#v, want merged/normalized capabilities", normalizeSpec)
|
||||||
|
}
|
||||||
|
codecSpec, ok := catalog.ArtifactCodecs.Spec(dnd.NPCListKind)
|
||||||
|
if !ok || codecSpec.Kind != dnd.NPCListKind || codecSpec.Schema.ID != npccodec.SchemaID || codecSpec.Schema.Version != npccodec.SchemaVersion {
|
||||||
|
t.Fatalf("NPC codec spec = %#v, want typed v1 durable schema", codecSpec)
|
||||||
|
}
|
||||||
|
|
||||||
|
wantExtractChain := []pipeline.ModuleBinding{
|
||||||
|
pipeline.Binding("generic/valid_json"),
|
||||||
|
pipeline.Binding("extract/dnd/npcs/shape"),
|
||||||
|
pipeline.Binding("extract/dnd/npcs/source_refs"),
|
||||||
|
pipeline.Binding("generic/valid_json_schema"),
|
||||||
|
pipeline.Binding("extract/dnd/npcs/source_relatedness"),
|
||||||
|
}
|
||||||
|
wantNormalizeChain := []pipeline.ModuleBinding{
|
||||||
|
pipeline.Binding("generic/valid_json"),
|
||||||
|
pipeline.Binding("extract/dnd/npcs/shape"),
|
||||||
|
pipeline.Binding("normalize/dnd/npcs/identity"),
|
||||||
|
pipeline.Binding("extract/dnd/npcs/source_refs"),
|
||||||
|
pipeline.Binding("generic/valid_json_schema"),
|
||||||
|
pipeline.Binding("extract/dnd/npcs/source_relatedness"),
|
||||||
|
}
|
||||||
|
if got := validatorChain(effective.ResolvedPipeline, pipeline.StageExtract, npcextract.Key); !reflect.DeepEqual(got, wantExtractChain) {
|
||||||
|
t.Fatalf("NPC extract chain = %#v, want %#v", got, wantExtractChain)
|
||||||
|
}
|
||||||
|
if got := validatorChain(effective.ResolvedPipeline, pipeline.StageNormalize, npcnormalize.Key); !reflect.DeepEqual(got, wantNormalizeChain) {
|
||||||
|
t.Fatalf("NPC normalize chain = %#v, want %#v", got, wantNormalizeChain)
|
||||||
|
}
|
||||||
|
if got := validatorChain(effective.ResolvedPipeline, pipeline.StageMerge, pipeline.DefaultMergeModule); len(got) != 0 {
|
||||||
|
t.Fatalf("NPC merge chain = %#v, want empty", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestProductionNPCConfigurationValidatesOptionsReferencesAndPlacement(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
resolve := func(mutate func(*pipeline.PipelineProfile)) error {
|
||||||
|
cfg := productionNPCContractConfig()
|
||||||
|
profile := cfg.Pipelines["dnd-session"]
|
||||||
|
mutate(&profile)
|
||||||
|
cfg.Pipelines["dnd-session"] = profile
|
||||||
|
_, err := cfg.Resolve(config.ResolveInput{PipelineID: "dnd-session", Catalog: catalogFromRegistries(components.registries)})
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := resolve(func(profile *pipeline.PipelineProfile) {
|
||||||
|
lane := profile.Artifacts["npcs"]
|
||||||
|
lane.Extract.Options = map[string]any{"unexpected": true}
|
||||||
|
profile.Artifacts["npcs"] = lane
|
||||||
|
}); err == nil || !strings.Contains(err.Error(), "unknown option") {
|
||||||
|
t.Fatalf("unknown extractor option error = %v, want strict option rejection", err)
|
||||||
|
}
|
||||||
|
if err := resolve(func(profile *pipeline.PipelineProfile) {
|
||||||
|
lane := profile.Artifacts["npcs"]
|
||||||
|
lane.Normalize.Options = map[string]any{"unexpected": true}
|
||||||
|
profile.Artifacts["npcs"] = lane
|
||||||
|
}); err == nil || !strings.Contains(err.Error(), "unknown option") {
|
||||||
|
t.Fatalf("unknown normalizer option error = %v, want strict option rejection", err)
|
||||||
|
}
|
||||||
|
if err := resolve(func(profile *pipeline.PipelineProfile) {
|
||||||
|
profile.References = pipeline.ExternalReferenceMap(map[string]string{
|
||||||
|
"players": "players.txt",
|
||||||
|
"party": "party.txt",
|
||||||
|
"glossary": "glossary.txt",
|
||||||
|
})
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatalf("optional NPC references error = %v, want resolution success", err)
|
||||||
|
}
|
||||||
|
if err := resolve(func(profile *pipeline.PipelineProfile) {
|
||||||
|
lane := profile.Artifacts["npcs"]
|
||||||
|
lane.Validators = []pipeline.ModuleBinding{pipeline.Binding("normalize/dnd/npcs/identity")}
|
||||||
|
profile.Artifacts["npcs"] = lane
|
||||||
|
}); err == nil || !strings.Contains(err.Error(), "artifact lane level") {
|
||||||
|
t.Fatalf("lane-level validator error = %v, want invalid placement rejection", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func productionNPCContractConfig() config.Config {
|
||||||
|
cfg := config.Default()
|
||||||
|
cfg.Pipelines["dnd-session"] = pipeline.PipelineProfile{
|
||||||
|
ID: "dnd-session",
|
||||||
|
Input: pipeline.Binding("seriatim"),
|
||||||
|
Chunk: pipeline.Binding(pipeline.DefaultChunkModule),
|
||||||
|
Artifacts: map[string]pipeline.ArtifactLaneProfile{
|
||||||
|
"npcs": {
|
||||||
|
Extract: pipeline.ModuleBinding{Module: npcextract.Key, Retries: 2},
|
||||||
|
Normalize: pipeline.Binding(npcnormalize.Key),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
return cfg
|
||||||
|
}
|
||||||
|
|
||||||
|
func validatorChain(resolved pipeline.ResolvedPipeline, stage pipeline.ModuleStage, module string) []pipeline.ModuleBinding {
|
||||||
|
for _, chain := range resolved.ValidatorChains {
|
||||||
|
if chain.Stage == stage && chain.ModuleKey == module {
|
||||||
|
bindings := make([]pipeline.ModuleBinding, len(chain.Validators))
|
||||||
|
for index, validator := range chain.Validators {
|
||||||
|
bindings[index] = validator.Binding
|
||||||
|
}
|
||||||
|
return bindings
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
111
internal/cli/dnd_scene_descriptions_contract_test.go
Normal file
111
internal/cli/dnd_scene_descriptions_contract_test.go
Normal file
@@ -0,0 +1,111 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
|
"reflect"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||||
|
scenecodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/scenedescriptions"
|
||||||
|
sceneextract "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/scenedescriptions"
|
||||||
|
scenenormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/scenedescriptions"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestProductionSceneDescriptionWorkflow(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
cfg := config.Default()
|
||||||
|
cfg.Pipelines["scene-descriptions"] = pipeline.PipelineProfile{
|
||||||
|
ID: "scene-descriptions",
|
||||||
|
Input: pipeline.Binding("seriatim"),
|
||||||
|
Chunk: pipeline.ModuleBinding{Module: "generic", Options: map[string]any{"max_units": 1}},
|
||||||
|
Output: pipeline.Binding("json"),
|
||||||
|
Artifacts: map[string]pipeline.ArtifactLaneProfile{
|
||||||
|
"scene-descriptions": {
|
||||||
|
Extract: pipeline.Binding(sceneextract.Key),
|
||||||
|
Normalize: pipeline.Binding(scenenormalize.Key),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
effective, err := cfg.Resolve(config.ResolveInput{PipelineID: "scene-descriptions", Catalog: catalogFromRegistries(components.registries)})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Resolve() error = %v", err)
|
||||||
|
}
|
||||||
|
lane := effective.ResolvedPipeline.Steps[0].ArtifactLanes[0]
|
||||||
|
if lane.ArtifactKind != dnd.SceneDescriptionListKind || lane.Extract.Module != sceneextract.Key || lane.Merge.Module != pipeline.DefaultMergeModule || lane.Normalize.Module != scenenormalize.Key {
|
||||||
|
t.Fatalf("resolved lane = %#v, want production scene-description composition", lane)
|
||||||
|
}
|
||||||
|
if len(lane.ExtractReferences.Bindings) != 0 || len(lane.NormalizeReferences.Bindings) != 0 {
|
||||||
|
t.Fatalf("resolved references = %#v / %#v, want no generated or required references", lane.ExtractReferences, lane.NormalizeReferences)
|
||||||
|
}
|
||||||
|
|
||||||
|
prepared, err := pipeline.Prepare(effective.ResolvedPipeline, components.registries, pipeline.ModuleDependencies{LLM: sceneDescriptionLLM{}})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Prepare() error = %v", err)
|
||||||
|
}
|
||||||
|
output, err := pipeline.New().Run(context.Background(), pipeline.RunInput{
|
||||||
|
Prepared: prepared,
|
||||||
|
RawInput: readRepositoryFile(t, "examples", "seriatim-minimal-transcript.json"),
|
||||||
|
ChunkCacheMode: pipeline.ChunkCacheBypass,
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Run() error = %v", err)
|
||||||
|
}
|
||||||
|
if output.Manifest.ValidationStatus != "approved" || len(output.Rejected) != 0 || len(output.NormalizeOutputs) != 1 {
|
||||||
|
t.Fatalf("run output = %#v, want one approved normalized artifact", output)
|
||||||
|
}
|
||||||
|
normalizedOutput := output.NormalizeOutputs[0]
|
||||||
|
if normalizedOutput.NormalizerKey != scenenormalize.Key || normalizedOutput.Artifact.Kind != dnd.SceneDescriptionListKind || normalizedOutput.Artifact.Schema.ID != scenecodec.SchemaID || normalizedOutput.Artifact.Schema.Name != scenecodec.SchemaName || normalizedOutput.Artifact.Schema.Version != scenecodec.SchemaVersion {
|
||||||
|
t.Fatalf("normalized output = %#v, want registered durable scene-description schema", normalizedOutput)
|
||||||
|
}
|
||||||
|
|
||||||
|
var value dnd.SceneDescriptionList
|
||||||
|
if err := json.Unmarshal(normalizedOutput.Artifact.Content, &value); err != nil {
|
||||||
|
t.Fatalf("decode normalized artifact: %v", err)
|
||||||
|
}
|
||||||
|
want := dnd.SceneDescriptionList{Scenes: []dnd.SceneDescription{
|
||||||
|
{ID: "chunk-000001", SourceRef: source.SourceRef{SourceID: "session-alpha", StartUnitID: 1, EndUnitID: 1}, Kind: dnd.SceneKindNarrative, Title: "Aria casts Cure Wounds", Summary: "Aria casts Cure Wounds."},
|
||||||
|
{ID: "chunk-000002", SourceRef: source.SourceRef{SourceID: "session-alpha", StartUnitID: 2, EndUnitID: 2}, Kind: dnd.SceneKindCombat, Title: "Bandit mage casts Shield", Summary: "The bandit mage casts Shield."},
|
||||||
|
}}
|
||||||
|
if !reflect.DeepEqual(value, want) {
|
||||||
|
t.Fatalf("normalized scene descriptions = %#v, want %#v", value, want)
|
||||||
|
}
|
||||||
|
durable := decodeAssembledOutput[dnd.SceneDescriptionList](t, output.OutputFiles, "lanes/scene-descriptions.json")
|
||||||
|
if !reflect.DeepEqual(durable, want) {
|
||||||
|
t.Fatalf("durable output payload = %#v, want %#v", durable, want)
|
||||||
|
}
|
||||||
|
if len(output.Warnings) != 0 {
|
||||||
|
t.Fatalf("warnings = %#v, want grounded descriptions without warnings", output.Warnings)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
type sceneDescriptionLLM struct{}
|
||||||
|
|
||||||
|
func (sceneDescriptionLLM) CompleteStructured(ctx context.Context, req contracts.StructuredCompletionRequest, out any) (contracts.StructuredCompletionResponse, error) {
|
||||||
|
if err := ctx.Err(); err != nil {
|
||||||
|
return contracts.StructuredCompletionResponse{}, err
|
||||||
|
}
|
||||||
|
if req.PromptID != sceneextract.PromptID {
|
||||||
|
return contracts.StructuredCompletionResponse{}, fmt.Errorf("unexpected prompt %q", req.PromptID)
|
||||||
|
}
|
||||||
|
transcript := string(req.Inputs["transcript"].Content)
|
||||||
|
var content string
|
||||||
|
switch {
|
||||||
|
case strings.Contains(transcript, "Cure Wounds"):
|
||||||
|
content = `{"kind":"narrative","title":" Aria casts Cure Wounds ","summary":" Aria casts Cure Wounds. "}`
|
||||||
|
case strings.Contains(transcript, "Shield"):
|
||||||
|
content = `{"kind":"combat","title":"Bandit mage casts Shield","summary":"The bandit mage casts Shield."}`
|
||||||
|
default:
|
||||||
|
return contracts.StructuredCompletionResponse{}, fmt.Errorf("unexpected transcript material %q", transcript)
|
||||||
|
}
|
||||||
|
if err := json.Unmarshal([]byte(content), out); err != nil {
|
||||||
|
return contracts.StructuredCompletionResponse{}, fmt.Errorf("populate structured response: %w", err)
|
||||||
|
}
|
||||||
|
return contracts.StructuredCompletionResponse{Content: []byte(content), Provider: "test", Model: "deterministic", ProfileID: req.ProfileID}, nil
|
||||||
|
}
|
||||||
235
internal/cli/example_contract_test.go
Normal file
235
internal/cli/example_contract_test.go
Normal file
@@ -0,0 +1,235 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"sort"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/debugbundle"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||||
|
spellnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/spells"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/modules/seriatim/input/transcript"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestMaintainedExamplesLoadResolveAndList(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
for _, example := range maintainedExampleFiles(t) {
|
||||||
|
t.Run(example.name, func(t *testing.T) {
|
||||||
|
cfg := loadMaintainedExample(t, example.path)
|
||||||
|
raw, err := os.ReadFile(example.transcriptPath)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read maintained transcript %q: %v", example.transcriptPath, err)
|
||||||
|
}
|
||||||
|
document, err := transcript.New().Parse(context.Background(), contracts.ParseRequest{
|
||||||
|
Path: example.transcriptPath,
|
||||||
|
Raw: raw,
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse maintained transcript %q: %v", example.transcriptPath, err)
|
||||||
|
}
|
||||||
|
if len(document.Units) == 0 {
|
||||||
|
t.Fatalf("maintained transcript %q has no parsed units", example.transcriptPath)
|
||||||
|
}
|
||||||
|
for _, pipelineID := range example.pipelineIDs {
|
||||||
|
effective, err := cfg.Resolve(resolveInputForMaintainedExample(components, pipelineID))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("resolve maintained example %q: %v", pipelineID, err)
|
||||||
|
}
|
||||||
|
materialized, _, err := pipeline.MaterializeReferences(effective.ResolvedPipeline, catalogFromRegistries(components.registries), pipeline.ReferenceMaterializationOptions{
|
||||||
|
ConfigPath: example.path,
|
||||||
|
WorkingDir: filepath.Dir(example.path),
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("materialize maintained example references for %q: %v", pipelineID, err)
|
||||||
|
}
|
||||||
|
if example.name == "complete" {
|
||||||
|
if got := exampleStepLaneIDs(materialized); strings.Join(got, "|") != "describe-session:item-events,npcs,scene-descriptions|extract-events:combat-turns,npc-interactions,spells" {
|
||||||
|
t.Fatalf("complete example steps and lanes = %v, want every D&D extractor in the documented two-step composition", got)
|
||||||
|
}
|
||||||
|
spellLane := referenceContractLane(t, materialized, "spells")
|
||||||
|
if len(spellLane.ExtractReferences.ReferenceSet.Slots["spell_catalog"].Items) != 1 ||
|
||||||
|
len(spellLane.NormalizeReferences.ReferenceSet.Slots["spell_catalog"].Items) != 1 {
|
||||||
|
t.Fatalf("complete example spell catalog reference was not materialized: %#v", spellLane)
|
||||||
|
}
|
||||||
|
itemEventLane := referenceContractLane(t, materialized, "item-events")
|
||||||
|
for _, references := range []pipeline.ResolvedReferenceTarget{itemEventLane.ExtractReferences, itemEventLane.NormalizeReferences} {
|
||||||
|
if _, found := references.ReferenceSet.Slots["npcs"]; found {
|
||||||
|
t.Fatalf("item event lane unexpectedly depends on generated NPCs: %#v", itemEventLane)
|
||||||
|
}
|
||||||
|
if _, found := references.ReferenceSet.Slots["scene_descriptions"]; found {
|
||||||
|
t.Fatalf("item event lane unexpectedly depends on generated scene descriptions: %#v", itemEventLane)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
var stdout, stderr strings.Builder
|
||||||
|
code := RunWithOptions([]string{"pipelines", "list", "--config", example.path}, &stdout, &stderr, productionOptionsFromComponents(components))
|
||||||
|
if code != 0 || stdout.String() != strings.Join(example.pipelineIDs, "\n")+"\n" || stderr.Len() != 0 {
|
||||||
|
t.Fatalf("pipelines list: code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestMaintainedConfigurationExampleSet(t *testing.T) {
|
||||||
|
entries, err := os.ReadDir(repositoryPath("examples"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var names []string
|
||||||
|
for _, entry := range entries {
|
||||||
|
if !entry.IsDir() && strings.HasSuffix(entry.Name(), ".config.yml") {
|
||||||
|
names = append(names, entry.Name())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
sort.Strings(names)
|
||||||
|
if got := strings.Join(names, ","); got != "dnd-complete.config.yml,dnd-minimal.config.yml" {
|
||||||
|
t.Fatalf("maintained configuration examples = %q, want only the minimal and complete D&D examples", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func exampleStepLaneIDs(resolved pipeline.ResolvedPipeline) []string {
|
||||||
|
result := make([]string, 0, len(resolved.Steps))
|
||||||
|
for _, step := range resolved.Steps {
|
||||||
|
laneIDs := make([]string, 0, len(step.ArtifactLanes))
|
||||||
|
for _, lane := range step.ArtifactLanes {
|
||||||
|
laneIDs = append(laneIDs, lane.ID)
|
||||||
|
}
|
||||||
|
sort.Strings(laneIDs)
|
||||||
|
result = append(result, step.ID+":"+strings.Join(laneIDs, ","))
|
||||||
|
}
|
||||||
|
return result
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestMaintainedMinimalInvocationProducesJSONBundle(t *testing.T) {
|
||||||
|
outputRoot := filepath.Join(t.TempDir(), "output")
|
||||||
|
fake := &productionFakeLLMClient{}
|
||||||
|
options := productionRunOptions(t, fake)
|
||||||
|
var stdout, stderr strings.Builder
|
||||||
|
code := RunWithOptions([]string{
|
||||||
|
"run", "dnd-session",
|
||||||
|
"--config", repositoryPath("examples", "dnd-minimal.config.yml"),
|
||||||
|
"--input", repositoryPath("examples", "seriatim-minimal-transcript.json"),
|
||||||
|
"--only", "spells", "--chunk_cache", "bypass", "--output-dir", outputRoot,
|
||||||
|
}, &stdout, &stderr, options)
|
||||||
|
if code != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
if !strings.Contains(stdout.String(), `pipeline "dnd-session"`) || !strings.Contains(stdout.String(), "outputs=1 rejected=0") {
|
||||||
|
t.Fatalf("stdout=%q, want completed pipeline and counts", stdout.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
runRoot := filepath.Join(outputRoot, productionRunID)
|
||||||
|
index := readProductionJSON[exampleOutputIndex](t, filepath.Join(runRoot, "index.json"))
|
||||||
|
if index.ManifestFile != "manifest.json" || index.RejectedFile != "rejected.json" || index.WarningsFile != "warnings.json" || len(index.OutputFiles) != 1 {
|
||||||
|
t.Fatalf("index = %#v, want one spells output and fixed companion files", index)
|
||||||
|
}
|
||||||
|
entry := index.OutputFiles[0]
|
||||||
|
if entry.LaneID != "spells" || entry.File != "lanes/spells.json" || entry.MediaType != "application/json" || entry.SchemaID != "notarius.dnd.spells" || entry.SchemaVersion != "v1" {
|
||||||
|
t.Fatalf("index output entry = %#v, want spells JSON contract", entry)
|
||||||
|
}
|
||||||
|
|
||||||
|
manifest := readProductionJSON[artifacts.RunManifest](t, filepath.Join(runRoot, "manifest.json"))
|
||||||
|
if manifest.PipelineID != "dnd-session" || manifest.InputModule != "seriatim" || manifest.Chunker != "generic" || manifest.OutputEncoder != "json" || manifest.ValidationStatus != "approved" || manifest.ChunkPlan == nil || manifest.ChunkPlan.Action != "bypassed" {
|
||||||
|
t.Fatalf("manifest = %#v, want approved minimal run", manifest)
|
||||||
|
}
|
||||||
|
if len(manifest.ArtifactLanes) != 1 {
|
||||||
|
t.Fatalf("manifest lanes = %#v, want exactly spells", manifest.ArtifactLanes)
|
||||||
|
}
|
||||||
|
lane := manifest.ArtifactLanes[0]
|
||||||
|
if lane.ID != "spells" || lane.Extractor != "dnd/spells" || lane.Merger != "appendorder" || lane.Normalizer != spellnormalize.Key {
|
||||||
|
t.Fatalf("manifest lane = %#v, want production spells composition", lane)
|
||||||
|
}
|
||||||
|
if len(manifest.References) != 0 {
|
||||||
|
t.Fatalf("base-only manifest references = %#v, want no overlay provenance", manifest.References)
|
||||||
|
}
|
||||||
|
extractorMetadata, ok := lane.Metadata["extractor"].(map[string]any)
|
||||||
|
if !ok || len(stringValues(extractorMetadata["catalog_overlay_ids"])) != 0 {
|
||||||
|
t.Fatalf("base-only extractor metadata = %#v, want no overlay IDs", lane.Metadata)
|
||||||
|
}
|
||||||
|
|
||||||
|
artifact := readProductionJSON[dnd.SpellList](t, filepath.Join(runRoot, entry.File))
|
||||||
|
if len(artifact.SpellCasts) != 1 || artifact.SpellCasts[0].Spell != "Cure Wounds" || artifact.SpellCasts[0].SourceRefs[0].SourceID != "session-alpha" {
|
||||||
|
t.Fatalf("artifact = %#v, want one source-linked Cure Wounds cast", artifact)
|
||||||
|
}
|
||||||
|
rejected := readProductionJSON[struct {
|
||||||
|
Rejected []json.RawMessage `json:"rejected"`
|
||||||
|
}](t, filepath.Join(runRoot, "rejected.json"))
|
||||||
|
if len(rejected.Rejected) != 0 {
|
||||||
|
t.Fatalf("rejected = %#v, want empty rejection list", rejected.Rejected)
|
||||||
|
}
|
||||||
|
warnings := readProductionJSON[struct {
|
||||||
|
Warnings []json.RawMessage `json:"warnings"`
|
||||||
|
}](t, filepath.Join(runRoot, "warnings.json"))
|
||||||
|
if len(warnings.Warnings) != 0 {
|
||||||
|
t.Fatalf("warnings = %#v, want empty warning list", warnings.Warnings)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestMaintainedMalformedInputOnlyRecordsDebugFailureWhenRequested(t *testing.T) {
|
||||||
|
malformed := filepath.Join(t.TempDir(), "malformed.json")
|
||||||
|
if err := os.WriteFile(malformed, []byte("{not valid json"), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
for _, debug := range []bool{false, true} {
|
||||||
|
name := "without debug"
|
||||||
|
if debug {
|
||||||
|
name = "with debug"
|
||||||
|
}
|
||||||
|
t.Run(name, func(t *testing.T) {
|
||||||
|
outputRoot := filepath.Join(t.TempDir(), "output")
|
||||||
|
debugRoot := filepath.Join(t.TempDir(), "debug")
|
||||||
|
options := productionRunOptions(t, &productionFakeLLMClient{})
|
||||||
|
args := []string{
|
||||||
|
"run", "dnd-session",
|
||||||
|
"--config", repositoryPath("examples", "dnd-minimal.config.yml"),
|
||||||
|
"--input", malformed, "--chunk_cache", "bypass", "--output-dir", outputRoot,
|
||||||
|
}
|
||||||
|
if debug {
|
||||||
|
args = append(args, "--debug", "--debug-dir", debugRoot)
|
||||||
|
}
|
||||||
|
var stdout, stderr strings.Builder
|
||||||
|
code := RunWithOptions(args, &stdout, &stderr, options)
|
||||||
|
if code != 1 || stdout.Len() != 0 || !strings.Contains(stderr.String(), "parse input") {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
assertAbsent(t, outputRoot)
|
||||||
|
if !debug {
|
||||||
|
assertAbsent(t, debugRoot)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
bundle := onlyChildDir(t, debugRoot)
|
||||||
|
report := readProductionJSON[debugbundle.RunReport](t, filepath.Join(bundle, "summary", "run-report.json"))
|
||||||
|
if report.Succeeded || report.PipelineID != "dnd-session" {
|
||||||
|
t.Fatalf("failure report = %#v, want failed dnd-session report", report)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
type exampleOutputIndex struct {
|
||||||
|
ManifestFile string `json:"manifest_file"`
|
||||||
|
OutputFiles []exampleOutputIndexEntry `json:"output_files"`
|
||||||
|
RejectedFile string `json:"rejected_file"`
|
||||||
|
WarningsFile string `json:"warnings_file"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type exampleOutputIndexEntry struct {
|
||||||
|
LaneID string `json:"lane_id"`
|
||||||
|
MediaType string `json:"media_type"`
|
||||||
|
File string `json:"file"`
|
||||||
|
SchemaID string `json:"schema_id"`
|
||||||
|
SchemaVersion string `json:"schema_version"`
|
||||||
|
}
|
||||||
|
|
||||||
|
func resolveInputForMaintainedExample(components productionComponents, pipelineID string) config.ResolveInput {
|
||||||
|
return config.ResolveInput{PipelineID: pipelineID, Catalog: catalogFromRegistries(components.registries)}
|
||||||
|
}
|
||||||
80
internal/cli/npc_registry_contract_test.go
Normal file
80
internal/cli/npc_registry_contract_test.go
Normal file
@@ -0,0 +1,80 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
|
"io/fs"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestOversizedNPCRegistryFailsBeforeRuntimeAndCheckpointConstruction(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
npcPath := filepath.Join(t.TempDir(), "npcs.json")
|
||||||
|
if err := os.WriteFile(npcPath, []byte(strings.Repeat("x", 1048577)), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
checkpointRoot := filepath.Join(t.TempDir(), "checkpoints")
|
||||||
|
content := fmt.Sprintf(`version: 3
|
||||||
|
cache:
|
||||||
|
chunk_plans:
|
||||||
|
mode: bypass
|
||||||
|
checkpoints:
|
||||||
|
enabled: true
|
||||||
|
directory: %q
|
||||||
|
pipelines:
|
||||||
|
dnd-session:
|
||||||
|
input: seriatim
|
||||||
|
artifacts:
|
||||||
|
spells:
|
||||||
|
extract:
|
||||||
|
module: dnd/spells
|
||||||
|
references:
|
||||||
|
npcs: %q
|
||||||
|
normalize: dnd/spells
|
||||||
|
`, checkpointRoot, npcPath)
|
||||||
|
configPath := filepath.Join(t.TempDir(), "config.yml")
|
||||||
|
if err := os.WriteFile(configPath, []byte(content), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
llmConstructed := false
|
||||||
|
chunkStoreConstructed := false
|
||||||
|
options := Options{
|
||||||
|
Catalog: catalogFromRegistries(components.registries),
|
||||||
|
Registries: components.registries,
|
||||||
|
LLMClientFactory: func(context.Context, config.Config, string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||||
|
llmConstructed = true
|
||||||
|
return nil, nil, errors.New("LLM client must not be constructed")
|
||||||
|
},
|
||||||
|
ChunkPlanStoreFactory: func(string) (pipeline.ChunkPlanStore, error) {
|
||||||
|
chunkStoreConstructed = true
|
||||||
|
return nil, errors.New("chunk-plan store must not be constructed")
|
||||||
|
},
|
||||||
|
}
|
||||||
|
var stdout, stderr strings.Builder
|
||||||
|
code := RunWithOptions([]string{
|
||||||
|
"run", "dnd-session", "--config", configPath,
|
||||||
|
"--input", repositoryPath("examples", "seriatim-minimal-transcript.json"),
|
||||||
|
"--chunk_cache", "bypass", "--output-dir", t.TempDir(),
|
||||||
|
}, &stdout, &stderr, options)
|
||||||
|
for _, fragment := range []string{`pipeline "dnd-session"`, `reference slot "npcs"`, "1048577 bytes", "limit 1048576"} {
|
||||||
|
if code == 0 || !strings.Contains(stderr.String(), fragment) {
|
||||||
|
t.Fatalf("RunWithOptions() code = %d stderr = %q, want context fragment %q", code, stderr.String(), fragment)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if llmConstructed || chunkStoreConstructed {
|
||||||
|
t.Fatalf("runtime construction = LLM %t, chunk store %t; want materialization failure first", llmConstructed, chunkStoreConstructed)
|
||||||
|
}
|
||||||
|
if _, err := os.Stat(checkpointRoot); !errors.Is(err, fs.ErrNotExist) {
|
||||||
|
t.Fatalf("checkpoint root stat error = %v, want no checkpoint allocation", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
851
internal/cli/production_contract_test.go
Normal file
851
internal/cli/production_contract_test.go
Normal file
@@ -0,0 +1,851 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
|
"io/fs"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"reflect"
|
||||||
|
"runtime"
|
||||||
|
"sort"
|
||||||
|
"strings"
|
||||||
|
"sync"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/chunkmap"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/chunk/scenes"
|
||||||
|
combatcodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/combatturns"
|
||||||
|
itemeventcodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/itemevents"
|
||||||
|
spellcodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/spells"
|
||||||
|
combatextract "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/combatturns"
|
||||||
|
itemeventextract "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/itemevents"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/spells"
|
||||||
|
combatnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/combatturns"
|
||||||
|
itemeventnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/itemevents"
|
||||||
|
spellnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/spells"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/modules/generic/normalize/noop"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestProductionCatalogCoversMaintainedConfigurations(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
registries := components.registries
|
||||||
|
|
||||||
|
assertProductionContains(t, "inputs", registries.Inputs.RegisteredKeys(), []string{"seriatim"})
|
||||||
|
assertProductionContains(t, "chunkers", registries.Chunkers.RegisteredKeys(), []string{"dnd/scenes", "generic"})
|
||||||
|
assertProductionContains(t, "extractors", registries.Extractors.RegisteredKeys(), []string{"dnd/spells", "dnd/npcs", combatextract.Key, itemeventextract.Key})
|
||||||
|
assertProductionContains(t, "mergers", registries.Mergers.RegisteredKeys(), []string{"appendorder"})
|
||||||
|
assertProductionContains(t, "normalizers", registries.Normalizers.RegisteredKeys(), []string{"noop", spellnormalize.Key, "dnd/npcs", combatnormalize.Key, itemeventnormalize.Key})
|
||||||
|
assertProductionContains(t, "outputs", registries.Outputs.RegisteredKeys(), []string{"json"})
|
||||||
|
assertProductionContains(t, "validators", registries.Validators.RegisteredKeys(), []string{
|
||||||
|
"extract/dnd/spells/catalog",
|
||||||
|
"extract/dnd/spells/shape",
|
||||||
|
"extract/dnd/spells/source_refs",
|
||||||
|
"extract/dnd/spells/source_relatedness",
|
||||||
|
"extract/dnd/combat-turns/shape",
|
||||||
|
"extract/dnd/combat-turns/source_refs",
|
||||||
|
"extract/dnd/combat-turns/source_relatedness",
|
||||||
|
"normalize/dnd/combat-turns/invariants",
|
||||||
|
"extract/dnd/item-events/shape",
|
||||||
|
"extract/dnd/item-events/source_refs",
|
||||||
|
"extract/dnd/item-events/source_relatedness",
|
||||||
|
"normalize/dnd/item-events/invariants",
|
||||||
|
"generic/always_accept",
|
||||||
|
"generic/always_reject",
|
||||||
|
"generic/valid_json",
|
||||||
|
"generic/valid_json_schema",
|
||||||
|
})
|
||||||
|
assertProductionContains(t, "artifact codec kinds", registries.ArtifactCodecs.RegisteredKinds(), []contracts.ArtifactKind{dnd.SpellListKind, dnd.NPCListKind, dnd.CombatTurnListKind, dnd.ItemEventListKind})
|
||||||
|
assertProductionContains(t, "merger variants", registries.Mergers.RegisteredArtifactKinds(pipeline.DefaultMergeModule), []contracts.ArtifactKind{dnd.SpellListKind, dnd.NPCListKind, dnd.CombatTurnListKind, dnd.ItemEventListKind})
|
||||||
|
assertProductionContains(t, "normalizer variants", registries.Normalizers.RegisteredArtifactKinds(pipeline.DefaultNormalizeModule), []contracts.ArtifactKind{dnd.SpellListKind, dnd.NPCListKind, dnd.CombatTurnListKind, dnd.ItemEventListKind})
|
||||||
|
assertProductionContains(t, "spell normalizer variants", registries.Normalizers.RegisteredArtifactKinds(spellnormalize.Key), []contracts.ArtifactKind{dnd.SpellListKind})
|
||||||
|
assertProductionContains(t, "combat normalizer variants", registries.Normalizers.RegisteredArtifactKinds(combatnormalize.Key), []contracts.ArtifactKind{dnd.CombatTurnListKind})
|
||||||
|
assertProductionContains(t, "item event normalizer variants", registries.Normalizers.RegisteredArtifactKinds(itemeventnormalize.Key), []contracts.ArtifactKind{dnd.ItemEventListKind})
|
||||||
|
|
||||||
|
wantChain := []pipeline.ModuleBinding{
|
||||||
|
pipeline.Binding("generic/valid_json"),
|
||||||
|
pipeline.Binding("extract/dnd/spells/shape"),
|
||||||
|
pipeline.Binding("extract/dnd/spells/catalog"),
|
||||||
|
pipeline.Binding("extract/dnd/spells/source_refs"),
|
||||||
|
pipeline.Binding("generic/valid_json_schema"),
|
||||||
|
pipeline.Binding("extract/dnd/spells/source_relatedness"),
|
||||||
|
}
|
||||||
|
if got := registries.ValidatorChains.Validators(pipeline.StageExtract, spells.Key); !reflect.DeepEqual(got, wantChain) {
|
||||||
|
t.Fatalf("spell validator chain = %#v, want %#v", got, wantChain)
|
||||||
|
}
|
||||||
|
if got := registries.ValidatorChains.Validators(pipeline.StageNormalize, spellnormalize.Key); !reflect.DeepEqual(got, wantChain) {
|
||||||
|
t.Fatalf("spell normalize validator chain = %#v, want %#v", got, wantChain)
|
||||||
|
}
|
||||||
|
combatExtractChain := []pipeline.ModuleBinding{
|
||||||
|
pipeline.Binding("generic/valid_json"),
|
||||||
|
pipeline.Binding("extract/dnd/combat-turns/shape"),
|
||||||
|
pipeline.Binding("extract/dnd/combat-turns/source_refs"),
|
||||||
|
pipeline.Binding("generic/valid_json_schema"),
|
||||||
|
pipeline.Binding("extract/dnd/combat-turns/source_relatedness"),
|
||||||
|
}
|
||||||
|
combatNormalizeChain := []pipeline.ModuleBinding{
|
||||||
|
pipeline.Binding("generic/valid_json"),
|
||||||
|
pipeline.Binding("extract/dnd/combat-turns/shape"),
|
||||||
|
pipeline.Binding("normalize/dnd/combat-turns/invariants"),
|
||||||
|
pipeline.Binding("extract/dnd/combat-turns/source_refs"),
|
||||||
|
pipeline.Binding("generic/valid_json_schema"),
|
||||||
|
pipeline.Binding("extract/dnd/combat-turns/source_relatedness"),
|
||||||
|
}
|
||||||
|
if got := registries.ValidatorChains.Validators(pipeline.StageExtract, combatextract.Key); !reflect.DeepEqual(got, combatExtractChain) {
|
||||||
|
t.Fatalf("combat extract validator chain = %#v, want %#v", got, combatExtractChain)
|
||||||
|
}
|
||||||
|
if got := registries.ValidatorChains.Validators(pipeline.StageNormalize, combatnormalize.Key); !reflect.DeepEqual(got, combatNormalizeChain) {
|
||||||
|
t.Fatalf("combat normalize validator chain = %#v, want %#v", got, combatNormalizeChain)
|
||||||
|
}
|
||||||
|
itemEventExtractChain := []pipeline.ModuleBinding{
|
||||||
|
pipeline.Binding("generic/valid_json"),
|
||||||
|
pipeline.Binding("extract/dnd/item-events/shape"),
|
||||||
|
pipeline.Binding("extract/dnd/item-events/source_refs"),
|
||||||
|
pipeline.Binding("generic/valid_json_schema"),
|
||||||
|
pipeline.Binding("extract/dnd/item-events/source_relatedness"),
|
||||||
|
}
|
||||||
|
itemEventNormalizeChain := []pipeline.ModuleBinding{
|
||||||
|
pipeline.Binding("generic/valid_json"),
|
||||||
|
pipeline.Binding("extract/dnd/item-events/shape"),
|
||||||
|
pipeline.Binding("normalize/dnd/item-events/invariants"),
|
||||||
|
pipeline.Binding("extract/dnd/item-events/source_refs"),
|
||||||
|
pipeline.Binding("generic/valid_json_schema"),
|
||||||
|
pipeline.Binding("extract/dnd/item-events/source_relatedness"),
|
||||||
|
}
|
||||||
|
if got := registries.ValidatorChains.Validators(pipeline.StageExtract, itemeventextract.Key); !reflect.DeepEqual(got, itemEventExtractChain) {
|
||||||
|
t.Fatalf("item event extract validator chain = %#v, want %#v", got, itemEventExtractChain)
|
||||||
|
}
|
||||||
|
if got := registries.ValidatorChains.Validators(pipeline.StageNormalize, itemeventnormalize.Key); !reflect.DeepEqual(got, itemEventNormalizeChain) {
|
||||||
|
t.Fatalf("item event normalize validator chain = %#v, want %#v", got, itemEventNormalizeChain)
|
||||||
|
}
|
||||||
|
|
||||||
|
assetNames := productionAssetNames(t, components.assets.PromptFS)
|
||||||
|
requiredAssets := []string{
|
||||||
|
"dnd.scenes/dnd.scenes.yaml",
|
||||||
|
"dnd.scenes/instructions.md",
|
||||||
|
"dnd.scenes/sharedassets/common-dnd-references.md",
|
||||||
|
"dnd.scenes/sharedassets/common-dnd-system.md",
|
||||||
|
"dnd.scenes/sharedassets/common-dnd-transcript.md",
|
||||||
|
"dnd.scenes/task.md",
|
||||||
|
"dnd.spells/dnd.spells.yaml",
|
||||||
|
"dnd.spells/catalog.md",
|
||||||
|
"dnd.spells/instructions.md",
|
||||||
|
"dnd.spells/sharedassets/common-dnd-references.md",
|
||||||
|
"dnd.spells/sharedassets/common-dnd-system.md",
|
||||||
|
"dnd.spells/sharedassets/common-dnd-transcript.md",
|
||||||
|
"dnd.spells/task.md",
|
||||||
|
"dnd.combat_turns/dnd.combat_turns.yaml",
|
||||||
|
"dnd.combat_turns/instructions.md",
|
||||||
|
"dnd.combat_turns/sharedassets/common-dnd-references.md",
|
||||||
|
"dnd.combat_turns/sharedassets/common-dnd-system.md",
|
||||||
|
"dnd.combat_turns/sharedassets/common-dnd-transcript.md",
|
||||||
|
"dnd.combat_turns/task.md",
|
||||||
|
"dnd.item_events/dnd.item_events.yaml",
|
||||||
|
"dnd.item_events/instructions.md",
|
||||||
|
"dnd.item_events/sharedassets/common-dnd-extraction-evidence.md",
|
||||||
|
"dnd.item_events/sharedassets/common-dnd-identity.md",
|
||||||
|
"dnd.item_events/sharedassets/common-dnd-references.md",
|
||||||
|
"dnd.item_events/sharedassets/common-dnd-system.md",
|
||||||
|
"dnd.item_events/sharedassets/common-dnd-transcript.md",
|
||||||
|
"dnd.item_events/task.md",
|
||||||
|
}
|
||||||
|
assertProductionContains(t, "production prompt assets", assetNames, requiredAssets)
|
||||||
|
|
||||||
|
catalog := catalogFromRegistries(registries)
|
||||||
|
converted := registriesFromCatalog(catalog)
|
||||||
|
if converted.ArtifactCodecs != registries.ArtifactCodecs || converted.ArtifactEvidence != registries.ArtifactEvidence || converted.ValidatorChains != registries.ValidatorChains {
|
||||||
|
t.Fatal("catalog/registry conversion did not preserve artifact and validator registries")
|
||||||
|
}
|
||||||
|
codecSpec, ok := catalog.ArtifactCodecs.Spec(dnd.SpellListKind)
|
||||||
|
if !ok || codecSpec.Kind != dnd.SpellListKind || codecSpec.Schema.ID != spellcodec.SchemaID {
|
||||||
|
t.Fatalf("catalog codec spec = %#v, ok=%t, want typed D&D spell codec", codecSpec, ok)
|
||||||
|
}
|
||||||
|
combatCodecSpec, ok := catalog.ArtifactCodecs.Spec(dnd.CombatTurnListKind)
|
||||||
|
if !ok || combatCodecSpec.Kind != dnd.CombatTurnListKind || combatCodecSpec.Schema.ID != combatcodec.SchemaID {
|
||||||
|
t.Fatalf("combat codec spec = %#v, ok=%t, want typed D&D combat codec", combatCodecSpec, ok)
|
||||||
|
}
|
||||||
|
itemEventCodecSpec, ok := catalog.ArtifactCodecs.Spec(dnd.ItemEventListKind)
|
||||||
|
if !ok || itemEventCodecSpec.Kind != dnd.ItemEventListKind || itemEventCodecSpec.Schema.ID != itemeventcodec.SchemaID {
|
||||||
|
t.Fatalf("item event codec spec = %#v, ok=%t, want typed D&D item-event codec", itemEventCodecSpec, ok)
|
||||||
|
}
|
||||||
|
if got := catalog.ValidatorChains.Validators(pipeline.StageExtract, spells.Key); !reflect.DeepEqual(got, wantChain) {
|
||||||
|
t.Fatalf("catalog validator chain = %#v, want %#v", got, wantChain)
|
||||||
|
}
|
||||||
|
if got := catalog.ValidatorChains.Validators(pipeline.StageNormalize, spellnormalize.Key); !reflect.DeepEqual(got, wantChain) {
|
||||||
|
t.Fatalf("catalog spell normalize validator chain = %#v, want %#v", got, wantChain)
|
||||||
|
}
|
||||||
|
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDefaultCLICompositionValidatesRepresentativeConfiguration(t *testing.T) {
|
||||||
|
var stdout, stderr strings.Builder
|
||||||
|
code := RunWithOptions([]string{
|
||||||
|
"config", "validate", "--config", repositoryPath("examples", "dnd-minimal.config.yml"), "--pipeline", "dnd-session",
|
||||||
|
}, &stdout, &stderr, Options{})
|
||||||
|
if code != 0 || stderr.Len() != 0 {
|
||||||
|
t.Fatalf("validate representative config with default composition: code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestProductionPromptAssetsPrepareWithoutProviderCredentials(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
cfg := config.Default()
|
||||||
|
cfg.Pipelines["dnd-scenes"] = pipeline.PipelineProfile{
|
||||||
|
ID: "dnd-scenes",
|
||||||
|
Input: pipeline.Binding("seriatim"),
|
||||||
|
Chunk: pipeline.Binding("dnd/scenes"),
|
||||||
|
Artifacts: map[string]pipeline.ArtifactLaneProfile{
|
||||||
|
"spells": {Extract: pipeline.Binding("dnd/spells")},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
effective, err := cfg.Resolve(config.ResolveInput{PipelineID: "dnd-scenes", Catalog: catalogFromRegistries(components.registries)})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("resolve production scene pipeline: %v", err)
|
||||||
|
}
|
||||||
|
if _, err := pipeline.Prepare(effective.ResolvedPipeline, components.registries, pipeline.ModuleDependencies{LLM: &productionFakeLLMClient{}}); err != nil {
|
||||||
|
t.Fatalf("prepare production scene and spell modules: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestProductionSpellValidatorsPrepareFromMaterializedCatalog(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
configPath := writeProductionSpellCatalogContractConfig(t)
|
||||||
|
effective, err := loadMaintainedExample(t, configPath).Resolve(resolveInputForMaintainedExample(components, "dnd-session"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("resolve production spell configuration: %v", err)
|
||||||
|
}
|
||||||
|
materialized, _, err := pipeline.MaterializeReferences(effective.ResolvedPipeline, catalogFromRegistries(components.registries), pipeline.ReferenceMaterializationOptions{
|
||||||
|
ConfigPath: configPath,
|
||||||
|
WorkingDir: filepath.Dir(configPath),
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("materialize production spell references: %v", err)
|
||||||
|
}
|
||||||
|
extractItems := materialized.Steps[0].ArtifactLanes[0].ExtractReferences.ReferenceSet.Slots["spell_catalog"].Items
|
||||||
|
normalizeItems := materialized.Steps[0].ArtifactLanes[0].NormalizeReferences.ReferenceSet.Slots["spell_catalog"].Items
|
||||||
|
if len(extractItems) != 1 || extractItems[0].MediaType != "application/json" || len(extractItems[0].Content) == 0 {
|
||||||
|
t.Fatalf("materialized extract spell catalog items = %#v, want one JSON item", extractItems)
|
||||||
|
}
|
||||||
|
if len(normalizeItems) != 1 || normalizeItems[0].MediaType != "application/json" || !reflect.DeepEqual(normalizeItems[0].Content, extractItems[0].Content) {
|
||||||
|
t.Fatalf("materialized normalize spell catalog items = %#v, want an independent binding of the extract catalog", normalizeItems)
|
||||||
|
}
|
||||||
|
if _, err := pipeline.Prepare(materialized, components.registries, pipeline.ModuleDependencies{LLM: &productionFakeLLMClient{}}); err != nil {
|
||||||
|
t.Fatalf("prepare production spell pipeline from materialized catalog: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestProductionSpellNormalizerRejectsInvalidCatalogReferencesBeforeExecution(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
configPath := writeProductionSpellCatalogContractConfig(t)
|
||||||
|
resolve := func(t *testing.T) pipeline.ResolvedPipeline {
|
||||||
|
t.Helper()
|
||||||
|
effective, err := loadMaintainedExample(t, configPath).Resolve(resolveInputForMaintainedExample(components, "dnd-session"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("resolve production spell configuration: %v", err)
|
||||||
|
}
|
||||||
|
return effective.ResolvedPipeline
|
||||||
|
}
|
||||||
|
materialize := func(resolved pipeline.ResolvedPipeline) (pipeline.ResolvedPipeline, error) {
|
||||||
|
materialized, _, err := pipeline.MaterializeReferences(resolved, catalogFromRegistries(components.registries), pipeline.ReferenceMaterializationOptions{
|
||||||
|
ConfigPath: configPath,
|
||||||
|
WorkingDir: filepath.Dir(configPath),
|
||||||
|
})
|
||||||
|
return materialized, err
|
||||||
|
}
|
||||||
|
|
||||||
|
t.Run("malformed catalog fails preparation", func(t *testing.T) {
|
||||||
|
catalogPath := filepath.Join(t.TempDir(), "malformed.json")
|
||||||
|
if err := os.WriteFile(catalogPath, []byte(`{"schema_version":`), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
resolved := resolve(t)
|
||||||
|
setNormalizeSpellCatalogSource(t, &resolved, catalogPath)
|
||||||
|
materialized, err := materialize(resolved)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("MaterializeReferences() error = %v, want malformed JSON to reach preparation", err)
|
||||||
|
}
|
||||||
|
_, err = pipeline.Prepare(materialized, components.registries, pipeline.ModuleDependencies{LLM: &productionFakeLLMClient{}})
|
||||||
|
for _, fragment := range []string{`pipeline "dnd-session"`, `lane "spells"`, "normalize", `module "dnd/spells"`, "decode spell catalog overlay"} {
|
||||||
|
if err == nil || !strings.Contains(err.Error(), fragment) {
|
||||||
|
t.Fatalf("Prepare() error = %v, want context fragment %q", err, fragment)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("multiple catalog items fail preparation", func(t *testing.T) {
|
||||||
|
materialized, err := materialize(resolve(t))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
slot := materialized.Steps[0].ArtifactLanes[0].NormalizeReferences.ReferenceSet.Slots["spell_catalog"]
|
||||||
|
slot.Items = append(slot.Items, slot.Items[0])
|
||||||
|
materialized.Steps[0].ArtifactLanes[0].NormalizeReferences.ReferenceSet.Slots["spell_catalog"] = slot
|
||||||
|
_, err = pipeline.Prepare(materialized, components.registries, pipeline.ModuleDependencies{LLM: &productionFakeLLMClient{}})
|
||||||
|
for _, fragment := range []string{"normalize", `module "dnd/spells"`, "zero or one item"} {
|
||||||
|
if err == nil || !strings.Contains(err.Error(), fragment) {
|
||||||
|
t.Fatalf("Prepare() error = %v, want context fragment %q", err, fragment)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("oversized catalog fails materialization", func(t *testing.T) {
|
||||||
|
catalogPath := filepath.Join(t.TempDir(), "oversized.json")
|
||||||
|
if err := os.WriteFile(catalogPath, []byte(strings.Repeat("x", 1048577)), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
checkpointRoot := filepath.Join(t.TempDir(), "checkpoints")
|
||||||
|
content := productionSpellCatalogContractConfig(t)
|
||||||
|
catalogSource := repositoryPath("examples", "dnd-spell-catalog.json")
|
||||||
|
if count := strings.Count(content, catalogSource); count != 2 {
|
||||||
|
t.Fatalf("spell catalog source occurs %d times, want extract and normalize bindings", count)
|
||||||
|
}
|
||||||
|
content = strings.Replace(content, catalogSource, "__extract_catalog__", 1)
|
||||||
|
content = replaceRequiredOnce(t, content, catalogSource, catalogPath)
|
||||||
|
content = replaceRequiredOnce(t, content, "__extract_catalog__", catalogSource)
|
||||||
|
content = replaceRequiredOnce(t, content, " enabled: false\n directory: \"\"", " enabled: true\n directory: "+checkpointRoot)
|
||||||
|
configFile := filepath.Join(t.TempDir(), "config.yml")
|
||||||
|
if err := os.WriteFile(configFile, []byte(content), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
llmConstructed := false
|
||||||
|
chunkStoreConstructed := false
|
||||||
|
options := Options{
|
||||||
|
Catalog: catalogFromRegistries(components.registries),
|
||||||
|
Registries: components.registries,
|
||||||
|
LLMClientFactory: func(context.Context, config.Config, string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||||
|
llmConstructed = true
|
||||||
|
return nil, nil, errors.New("LLM client must not be constructed")
|
||||||
|
},
|
||||||
|
ChunkPlanStoreFactory: func(string) (pipeline.ChunkPlanStore, error) {
|
||||||
|
chunkStoreConstructed = true
|
||||||
|
return nil, errors.New("chunk-plan store must not be constructed")
|
||||||
|
},
|
||||||
|
}
|
||||||
|
var stdout, stderr strings.Builder
|
||||||
|
code := RunWithOptions([]string{
|
||||||
|
"run", "dnd-session", "--config", configFile,
|
||||||
|
"--input", repositoryPath("examples", "seriatim-minimal-transcript.json"),
|
||||||
|
}, &stdout, &stderr, options)
|
||||||
|
errText := stderr.String()
|
||||||
|
for _, fragment := range []string{"normalize", `lane "spells"`, `reference slot "spell_catalog"`, "1048577 bytes", "limit 1048576"} {
|
||||||
|
if code == 0 || !strings.Contains(errText, fragment) {
|
||||||
|
t.Fatalf("RunWithOptions() code = %d stderr = %q, want context fragment %q", code, errText, fragment)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if llmConstructed || chunkStoreConstructed {
|
||||||
|
t.Fatalf("runtime construction = LLM %t, chunk store %t; want materialization failure first", llmConstructed, chunkStoreConstructed)
|
||||||
|
}
|
||||||
|
if _, err := os.Stat(checkpointRoot); !errors.Is(err, fs.ErrNotExist) {
|
||||||
|
t.Fatalf("checkpoint root stat error = %v, want no checkpoint allocation", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func setNormalizeSpellCatalogSource(t *testing.T, resolved *pipeline.ResolvedPipeline, sourcePath string) {
|
||||||
|
t.Helper()
|
||||||
|
if resolved == nil || len(resolved.Steps[0].ArtifactLanes) != 1 {
|
||||||
|
t.Fatalf("resolved pipeline = %#v, want one artifact lane", resolved)
|
||||||
|
}
|
||||||
|
bindings := resolved.Steps[0].ArtifactLanes[0].NormalizeReferences.Bindings
|
||||||
|
matches := 0
|
||||||
|
for index := range bindings {
|
||||||
|
if bindings[index].SlotName == "spell_catalog" {
|
||||||
|
bindings[index].Source = sourcePath
|
||||||
|
matches++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if matches != 1 {
|
||||||
|
t.Fatalf("normalize reference bindings = %#v, want exactly one spell_catalog binding", bindings)
|
||||||
|
}
|
||||||
|
resolved.Steps[0].ArtifactLanes[0].NormalizeReferences.Bindings = bindings
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestProductionLLMClientFactoriesBuildOfflineRuntime(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
factories := []struct {
|
||||||
|
name string
|
||||||
|
factory LLMClientFactory
|
||||||
|
}{
|
||||||
|
{name: "default production assets", factory: productionLLMClientFactory},
|
||||||
|
{name: "provided production assets", factory: productionLLMClientFactoryWithAssets(components.assets)},
|
||||||
|
}
|
||||||
|
for _, tt := range factories {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
client, manifests, err := tt.factory(context.Background(), config.Default(), "test-profile")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("build production LLM runtime: %v", err)
|
||||||
|
}
|
||||||
|
if client == nil {
|
||||||
|
t.Fatal("production LLM runtime returned a nil client")
|
||||||
|
}
|
||||||
|
if len(manifests) != 0 {
|
||||||
|
t.Fatalf("eager profile manifests = %#v, want none", manifests)
|
||||||
|
}
|
||||||
|
if _, ok := client.(contracts.LLMProfileManifestProvider); !ok {
|
||||||
|
t.Fatalf("production LLM client %T does not provide profile manifests", client)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestProductionLLMClientFactoriesRejectInvalidConstruction(t *testing.T) {
|
||||||
|
t.Run("canceled context", func(t *testing.T) {
|
||||||
|
ctx, cancel := context.WithCancel(context.Background())
|
||||||
|
cancel()
|
||||||
|
client, manifests, err := productionLLMClientFactory(ctx, config.Default(), "test-profile")
|
||||||
|
if !errors.Is(err, context.Canceled) || client != nil || len(manifests) != 0 {
|
||||||
|
t.Fatalf("client=%T manifests=%#v error=%v, want canceled construction", client, manifests, err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("nil assets", func(t *testing.T) {
|
||||||
|
client, manifests, err := productionLLMClientFactoryWithAssets(nil)(context.Background(), config.Default(), "test-profile")
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "asset registry must not be nil") || client != nil || len(manifests) != 0 {
|
||||||
|
t.Fatalf("client=%T manifests=%#v error=%v, want nil-assets failure", client, manifests, err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("invalid scheduler concurrency", func(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
cfg := config.Default()
|
||||||
|
cfg.Concurrency.TotalLLM = 0
|
||||||
|
client, manifests, err := productionLLMClientFactoryWithAssets(components.assets)(context.Background(), cfg, "test-profile")
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "create LLM scheduler") || !strings.Contains(err.Error(), "greater than zero") || client != nil || len(manifests) != 0 {
|
||||||
|
t.Fatalf("client=%T manifests=%#v error=%v, want scheduler-construction failure", client, manifests, err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestProductionConfigValidationCoversModuleAndVariantFailures(t *testing.T) {
|
||||||
|
base := string(readRepositoryFile(t, "examples", "dnd-minimal.config.yml"))
|
||||||
|
validPath := writeProductionContractConfig(t, base)
|
||||||
|
options := productionCLIOptions(t)
|
||||||
|
var stdout, stderr strings.Builder
|
||||||
|
if code := RunWithOptions([]string{"config", "validate", "--config", validPath, "--pipeline", "dnd-session"}, &stdout, &stderr, options); code != 0 {
|
||||||
|
t.Fatalf("valid production config: code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
content string
|
||||||
|
options Options
|
||||||
|
fragments []string
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
name: "unknown module",
|
||||||
|
content: replaceRequiredOnce(t, base, " input: seriatim\n", " input: missing/input\n"),
|
||||||
|
options: productionCLIOptions(t),
|
||||||
|
fragments: []string{"pipeline \"dnd-session\"", "input", "missing/input"},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "unknown validator",
|
||||||
|
content: replaceRequiredOnce(t, base, " extract: dnd/spells\n", " extract:\n module: dnd/spells\n validators:\n - module: missing/validator\n"),
|
||||||
|
options: productionCLIOptions(t),
|
||||||
|
fragments: []string{"validator", "missing/validator"},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "invalid artifact variant",
|
||||||
|
content: base,
|
||||||
|
options: productionCLIOptionsWithoutSpellNormalizer(t),
|
||||||
|
fragments: []string{"normalizer", spellnormalize.Key, string(dnd.SpellListKind), "variant"},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "deterministic validator with profile",
|
||||||
|
content: replaceRequiredOnce(t, base, " extract: dnd/spells\n", " extract:\n module: dnd/spells\n validators:\n - module: generic/valid_json\n llm_profile: forbidden-profile\n"),
|
||||||
|
options: productionCLIOptions(t),
|
||||||
|
fragments: []string{"deterministic validator", "llm_profile"},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
path := writeProductionContractConfig(t, tt.content)
|
||||||
|
var stdout, stderr strings.Builder
|
||||||
|
code := RunWithOptions([]string{"config", "validate", "--config", path, "--pipeline", "dnd-session"}, &stdout, &stderr, tt.options)
|
||||||
|
if code != 1 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
for _, fragment := range tt.fragments {
|
||||||
|
if !strings.Contains(stderr.String(), fragment) {
|
||||||
|
t.Fatalf("stderr=%q, want %q", stderr.String(), fragment)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestProductionNormalizeValidatorOverrideRemainsAuthoritative(t *testing.T) {
|
||||||
|
base := string(readRepositoryFile(t, "examples", "dnd-minimal.config.yml"))
|
||||||
|
content := replaceRequiredOnce(t, base, " normalize: dnd/spells\n", " normalize:\n module: dnd/spells\n validators:\n - module: generic/always_accept\n - module: generic/valid_json\n")
|
||||||
|
path := writeProductionContractConfig(t, content)
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
effective, err := loadMaintainedExample(t, path).Resolve(resolveInputForMaintainedExample(components, "dnd-session"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("resolve normalize override: %v", err)
|
||||||
|
}
|
||||||
|
for _, chain := range effective.ResolvedPipeline.ValidatorChains {
|
||||||
|
if chain.Stage != pipeline.StageNormalize || chain.ModuleKey != spellnormalize.Key {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if len(chain.Validators) != 2 || chain.Validators[0].Binding.Module != "generic/always_accept" || chain.Validators[1].Binding.Module != "generic/valid_json" {
|
||||||
|
t.Fatalf("normalize validator chain = %#v, want explicit validator order", chain)
|
||||||
|
}
|
||||||
|
return
|
||||||
|
}
|
||||||
|
t.Fatalf("resolved validator chains = %#v, want normalize chain for %q", effective.ResolvedPipeline.ValidatorChains, spellnormalize.Key)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestProductionSceneRunRecordsAnnotationFreeChunkPlanAndProvenance(t *testing.T) {
|
||||||
|
outputRoot := filepath.Join(t.TempDir(), "output")
|
||||||
|
configPath := writeProductionContractConfig(t, productionRunConfig(outputRoot, "dnd/scenes"))
|
||||||
|
fake := &productionFakeLLMClient{}
|
||||||
|
options := productionRunOptions(t, fake)
|
||||||
|
var stdout, stderr strings.Builder
|
||||||
|
code := RunWithOptions([]string{
|
||||||
|
"run", "dnd-session", "--config", configPath,
|
||||||
|
"--input", repositoryPath("examples", "seriatim-minimal-transcript.json"),
|
||||||
|
"--chunk_cache", "bypass", "--session-id", "offline-session",
|
||||||
|
}, &stdout, &stderr, options)
|
||||||
|
if code != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
manifest := readProductionJSON[artifacts.RunManifest](t, filepath.Join(outputRoot, productionRunID, "manifest.json"))
|
||||||
|
if manifest.Chunker != scenes.Key || manifest.ChunkPlan == nil || manifest.ChunkPlan.Action != "bypassed" || manifest.ChunkPlan.ProducerModule != scenes.Key {
|
||||||
|
t.Fatalf("chunk manifest = %#v, want dnd scene producer", manifest.ChunkPlan)
|
||||||
|
}
|
||||||
|
if got := manifest.ModuleMetadata["chunker"]["prompt_id"]; got != scenes.PromptID {
|
||||||
|
t.Fatalf("chunker prompt metadata = %#v, want %q", got, scenes.PromptID)
|
||||||
|
}
|
||||||
|
if got := manifest.ChunkPlan.ProducerMetadata["response_schema_id"]; got != scenes.ResponseSchemaID {
|
||||||
|
t.Fatalf("chunk producer schema metadata = %#v, want %q", got, scenes.ResponseSchemaID)
|
||||||
|
}
|
||||||
|
index := readProductionJSON[productionChunkMapIndex](t, filepath.Join(outputRoot, productionRunID, "index.json"))
|
||||||
|
if index.ChunkMap == nil || index.ChunkMap.ArtifactKind != chunkmap.ArtifactKind || index.ChunkMap.File != "chunk-map.json" || index.ChunkMap.MediaType != chunkmap.MediaType || index.ChunkMap.SchemaID != chunkmap.SchemaID || index.ChunkMap.SchemaName != chunkmap.SchemaName || index.ChunkMap.SchemaVersion != chunkmap.SchemaVersion {
|
||||||
|
t.Fatalf("chunk map index = %#v, want fixed chunk map descriptor", index.ChunkMap)
|
||||||
|
}
|
||||||
|
for _, output := range index.OutputFiles {
|
||||||
|
if output.File == index.ChunkMap.File {
|
||||||
|
t.Fatalf("lane output files = %#v, want no chunk map", index.OutputFiles)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
content, err := os.ReadFile(filepath.Join(outputRoot, productionRunID, index.ChunkMap.File))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
chunkMap, err := chunkmap.New().Decode(content)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Decode(chunk map) error = %v", err)
|
||||||
|
}
|
||||||
|
if chunkMap.SourceID != "session-alpha" || chunkMap.SourceDigest != manifest.ChunkPlan.SourceDigest || chunkMap.PlanDigest != manifest.ChunkPlan.PlanDigest || chunkMap.RequestedChunker != scenes.Key || chunkMap.Producer.InputModule != "seriatim" || chunkMap.Producer.ChunkModule != scenes.Key || chunkMap.Producer.LLMProfile != manifest.ChunkPlan.ProducerLLMProfile {
|
||||||
|
t.Fatalf("chunk map identity and producer = %#v, want accepted scene plan provenance", chunkMap)
|
||||||
|
}
|
||||||
|
if len(chunkMap.Chunks) != 1 || chunkMap.Chunks[0].ID != "chunk-000001" || chunkMap.Chunks[0].Index != 0 || chunkMap.Chunks[0].SourceRef.SourceID != "session-alpha" || chunkMap.Chunks[0].SourceRef.StartUnitID != 1 || chunkMap.Chunks[0].SourceRef.EndUnitID != 2 || chunkMap.Chunks[0].UnitCount != 2 {
|
||||||
|
t.Fatalf("chunk map chunks = %#v, want one stable accepted scene range", chunkMap.Chunks)
|
||||||
|
}
|
||||||
|
if len(chunkMap.PlanAnnotations) != 0 {
|
||||||
|
t.Fatalf("chunk map plan annotations = %#v, want none", chunkMap.PlanAnnotations)
|
||||||
|
}
|
||||||
|
if len(chunkMap.Chunks[0].Annotations) != 0 {
|
||||||
|
t.Fatalf("chunk map range annotations = %#v, want none", chunkMap.Chunks[0].Annotations)
|
||||||
|
}
|
||||||
|
warnings := readProductionJSON[struct {
|
||||||
|
Warnings []contracts.Warning `json:"warnings"`
|
||||||
|
}](t, filepath.Join(outputRoot, productionRunID, "warnings.json"))
|
||||||
|
if len(warnings.Warnings) != 0 {
|
||||||
|
t.Fatalf("warnings = %#v, want none", warnings.Warnings)
|
||||||
|
}
|
||||||
|
if len(fake.requestsFor(scenes.PromptID)) != 1 || len(fake.requestsFor(spells.PromptID)) != 1 || len(fake.requestsFor(itemeventextract.PromptID)) != 1 {
|
||||||
|
t.Fatalf("fake prompt requests = %#v, want one scene, spell, and item-event request", fake.requestPrompts())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
type maintainedExample struct {
|
||||||
|
name string
|
||||||
|
path string
|
||||||
|
transcriptPath string
|
||||||
|
pipelineIDs []string
|
||||||
|
}
|
||||||
|
|
||||||
|
func maintainedExampleFiles(t *testing.T) []maintainedExample {
|
||||||
|
t.Helper()
|
||||||
|
return []maintainedExample{
|
||||||
|
{name: "minimal", path: repositoryPath("examples", "dnd-minimal.config.yml"), transcriptPath: repositoryPath("examples", "seriatim-minimal-transcript.json"), pipelineIDs: []string{"dnd-session"}},
|
||||||
|
{name: "complete", path: repositoryPath("examples", "dnd-complete.config.yml"), transcriptPath: repositoryPath("examples", "dnd-complete-transcript.json"), pipelineIDs: []string{"dnd-session"}},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func productionSpellCatalogContractConfig(t *testing.T) string {
|
||||||
|
t.Helper()
|
||||||
|
return fmt.Sprintf(`version: 3
|
||||||
|
cache:
|
||||||
|
chunk_plans:
|
||||||
|
mode: bypass
|
||||||
|
checkpoints:
|
||||||
|
enabled: false
|
||||||
|
directory: ""
|
||||||
|
pipelines:
|
||||||
|
dnd-session:
|
||||||
|
input: seriatim
|
||||||
|
references:
|
||||||
|
party: %q
|
||||||
|
glossary: %q
|
||||||
|
artifacts:
|
||||||
|
spells:
|
||||||
|
extract:
|
||||||
|
module: dnd/spells
|
||||||
|
retries: 2
|
||||||
|
references:
|
||||||
|
spell_catalog: %q
|
||||||
|
normalize:
|
||||||
|
module: dnd/spells
|
||||||
|
references:
|
||||||
|
spell_catalog: %q
|
||||||
|
`, repositoryPath("examples", "dnd-party.txt"), repositoryPath("examples", "dnd-glossary.txt"), repositoryPath("examples", "dnd-spell-catalog.json"), repositoryPath("examples", "dnd-spell-catalog.json"))
|
||||||
|
}
|
||||||
|
|
||||||
|
func writeProductionSpellCatalogContractConfig(t *testing.T) string {
|
||||||
|
t.Helper()
|
||||||
|
return writeProductionContractConfig(t, productionSpellCatalogContractConfig(t))
|
||||||
|
}
|
||||||
|
|
||||||
|
func loadMaintainedExample(t *testing.T, path string) config.Config {
|
||||||
|
t.Helper()
|
||||||
|
fileConfig, err := config.LoadFileConfig(path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("load maintained config %q: %v", path, err)
|
||||||
|
}
|
||||||
|
cfg := config.Default()
|
||||||
|
if err := cfg.ApplyFileConfig(fileConfig); err != nil {
|
||||||
|
t.Fatalf("apply maintained config %q: %v", path, err)
|
||||||
|
}
|
||||||
|
if err := cfg.Validate(); err != nil {
|
||||||
|
t.Fatalf("validate maintained config %q: %v", path, err)
|
||||||
|
}
|
||||||
|
return cfg
|
||||||
|
}
|
||||||
|
|
||||||
|
func productionTestComponents(t *testing.T) productionComponents {
|
||||||
|
t.Helper()
|
||||||
|
components, err := newProductionComponents()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("new production components: %v", err)
|
||||||
|
}
|
||||||
|
return components
|
||||||
|
}
|
||||||
|
|
||||||
|
func productionCLIOptions(t *testing.T) Options {
|
||||||
|
t.Helper()
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
return productionOptionsFromComponents(components)
|
||||||
|
}
|
||||||
|
|
||||||
|
func productionOptionsFromComponents(components productionComponents) Options {
|
||||||
|
return Options{
|
||||||
|
Catalog: catalogFromRegistries(components.registries),
|
||||||
|
Registries: components.registries,
|
||||||
|
LookupEnv: emptyLookup,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func productionCLIOptionsWithoutSpellNormalizer(t *testing.T) Options {
|
||||||
|
t.Helper()
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
registries := components.registries
|
||||||
|
registries.Normalizers = pipeline.NewNormalizerRegistry()
|
||||||
|
if err := noop.RegisterTyped[dnd.SpellList](registries.Normalizers, contracts.ArtifactKind("test/other")); err != nil {
|
||||||
|
t.Fatalf("register mismatched normalizer: %v", err)
|
||||||
|
}
|
||||||
|
return productionOptionsFromComponents(productionComponents{registries: registries, assets: components.assets})
|
||||||
|
}
|
||||||
|
|
||||||
|
const productionRunID = "run-1700000000000000000-0123456789abcdef0123456789abcdef"
|
||||||
|
|
||||||
|
func productionRunOptions(t *testing.T, fake *productionFakeLLMClient) Options {
|
||||||
|
t.Helper()
|
||||||
|
options := productionCLIOptions(t)
|
||||||
|
options.Now = func() time.Time { return time.Unix(1700000000, 0).UTC() }
|
||||||
|
options.RunIDGenerator = func(time.Time) (string, error) { return productionRunID, nil }
|
||||||
|
options.UserCacheDir = func() (string, error) { return "", errors.New("user cache must not be used") }
|
||||||
|
options.LLMClientFactory = func(context.Context, config.Config, string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||||
|
return fake, nil, nil
|
||||||
|
}
|
||||||
|
return options
|
||||||
|
}
|
||||||
|
|
||||||
|
func productionRunConfig(outputRoot, chunkModule string) string {
|
||||||
|
return fmt.Sprintf(`version: 3
|
||||||
|
output:
|
||||||
|
directory: %q
|
||||||
|
cache:
|
||||||
|
chunk_plans:
|
||||||
|
mode: bypass
|
||||||
|
checkpoints: {}
|
||||||
|
debug:
|
||||||
|
directory: %q
|
||||||
|
pipelines:
|
||||||
|
dnd-session:
|
||||||
|
input: seriatim
|
||||||
|
chunk: %s
|
||||||
|
output:
|
||||||
|
module: json
|
||||||
|
options:
|
||||||
|
include_chunk_map: true
|
||||||
|
artifacts:
|
||||||
|
spells:
|
||||||
|
extract: dnd/spells
|
||||||
|
item-events:
|
||||||
|
extract: dnd/item-events
|
||||||
|
`, outputRoot, filepath.Join(filepath.Dir(outputRoot), "debug"), chunkModule)
|
||||||
|
}
|
||||||
|
|
||||||
|
type productionChunkMapIndex struct {
|
||||||
|
OutputFiles []struct {
|
||||||
|
File string `json:"file"`
|
||||||
|
} `json:"output_files"`
|
||||||
|
ChunkMap *struct {
|
||||||
|
ArtifactKind contracts.ArtifactKind `json:"artifact_kind"`
|
||||||
|
File string `json:"file"`
|
||||||
|
MediaType string `json:"media_type"`
|
||||||
|
SchemaID string `json:"schema_id"`
|
||||||
|
SchemaName string `json:"schema_name"`
|
||||||
|
SchemaVersion string `json:"schema_version"`
|
||||||
|
} `json:"chunk_map"`
|
||||||
|
}
|
||||||
|
|
||||||
|
func writeProductionContractConfig(t *testing.T, content string) string {
|
||||||
|
t.Helper()
|
||||||
|
path := filepath.Join(t.TempDir(), "config.yml")
|
||||||
|
if err := os.WriteFile(path, []byte(content), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return path
|
||||||
|
}
|
||||||
|
|
||||||
|
func productionAssetNames(t *testing.T, getFS func() (fs.FS, error)) []string {
|
||||||
|
t.Helper()
|
||||||
|
fileSystem, err := getFS()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("load production prompt assets: %v", err)
|
||||||
|
}
|
||||||
|
var names []string
|
||||||
|
if err := fs.WalkDir(fileSystem, ".", func(path string, entry fs.DirEntry, err error) error {
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if !entry.IsDir() {
|
||||||
|
names = append(names, path)
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatalf("walk production prompt assets: %v", err)
|
||||||
|
}
|
||||||
|
sort.Strings(names)
|
||||||
|
return names
|
||||||
|
}
|
||||||
|
|
||||||
|
func assertProductionContains[T comparable](t *testing.T, name string, got, required []T) {
|
||||||
|
t.Helper()
|
||||||
|
available := make(map[T]struct{}, len(got))
|
||||||
|
for _, entry := range got {
|
||||||
|
available[entry] = struct{}{}
|
||||||
|
}
|
||||||
|
var missing []T
|
||||||
|
for _, entry := range required {
|
||||||
|
if _, ok := available[entry]; !ok {
|
||||||
|
missing = append(missing, entry)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(missing) > 0 {
|
||||||
|
t.Fatalf("%s missing required entries %#v; registered entries are %#v", name, missing, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func readProductionJSON[T any](t *testing.T, path string) T {
|
||||||
|
t.Helper()
|
||||||
|
data, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read %s: %v", path, err)
|
||||||
|
}
|
||||||
|
var value T
|
||||||
|
if err := json.Unmarshal(data, &value); err != nil {
|
||||||
|
t.Fatalf("decode %s: %v", path, err)
|
||||||
|
}
|
||||||
|
return value
|
||||||
|
}
|
||||||
|
|
||||||
|
type productionFakeLLMClient struct {
|
||||||
|
mu sync.Mutex
|
||||||
|
requests []contracts.StructuredCompletionRequest
|
||||||
|
spellResponse string
|
||||||
|
}
|
||||||
|
|
||||||
|
func (client *productionFakeLLMClient) CompleteStructured(ctx context.Context, req contracts.StructuredCompletionRequest, out any) (contracts.StructuredCompletionResponse, error) {
|
||||||
|
if err := ctx.Err(); err != nil {
|
||||||
|
return contracts.StructuredCompletionResponse{}, err
|
||||||
|
}
|
||||||
|
var content []byte
|
||||||
|
switch req.PromptID {
|
||||||
|
case scenes.PromptID:
|
||||||
|
content = []byte(`{"scenes":[{"start_unit_id":1,"end_unit_id":2}]}`)
|
||||||
|
case spells.PromptID:
|
||||||
|
if client.spellResponse != "" {
|
||||||
|
content = []byte(client.spellResponse)
|
||||||
|
} else {
|
||||||
|
content = []byte(`{"spell_casts":[{"caster":"Aria","spell":"Cure Wounds","source_refs":[{"start_unit_id":1,"end_unit_id":1}]}]}`)
|
||||||
|
}
|
||||||
|
case itemeventextract.PromptID:
|
||||||
|
content = []byte(`{"events":[{"name":"Cure Wounds","kind":"acquired","to":"party","source_refs":[{"start_segment":1,"end_segment":1}]}]}`)
|
||||||
|
default:
|
||||||
|
return contracts.StructuredCompletionResponse{}, fmt.Errorf("unexpected prompt %q", req.PromptID)
|
||||||
|
}
|
||||||
|
if err := json.Unmarshal(content, out); err != nil {
|
||||||
|
return contracts.StructuredCompletionResponse{}, fmt.Errorf("populate fake structured target: %w", err)
|
||||||
|
}
|
||||||
|
client.mu.Lock()
|
||||||
|
client.requests = append(client.requests, req)
|
||||||
|
client.mu.Unlock()
|
||||||
|
return contracts.StructuredCompletionResponse{Content: content, Provider: "test", Model: "deterministic", ProfileID: req.ProfileID}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (client *productionFakeLLMClient) requestsFor(promptID string) []contracts.StructuredCompletionRequest {
|
||||||
|
client.mu.Lock()
|
||||||
|
defer client.mu.Unlock()
|
||||||
|
var requests []contracts.StructuredCompletionRequest
|
||||||
|
for _, req := range client.requests {
|
||||||
|
if req.PromptID == promptID {
|
||||||
|
requests = append(requests, req)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return requests
|
||||||
|
}
|
||||||
|
|
||||||
|
func (client *productionFakeLLMClient) requestPrompts() []string {
|
||||||
|
client.mu.Lock()
|
||||||
|
defer client.mu.Unlock()
|
||||||
|
prompts := make([]string, 0, len(client.requests))
|
||||||
|
for _, req := range client.requests {
|
||||||
|
prompts = append(prompts, req.PromptID)
|
||||||
|
}
|
||||||
|
return prompts
|
||||||
|
}
|
||||||
|
|
||||||
|
func repositoryPath(parts ...string) string {
|
||||||
|
_, file, _, _ := runtime.Caller(0)
|
||||||
|
return filepath.Join(append([]string{filepath.Dir(file), "..", ".."}, parts...)...)
|
||||||
|
}
|
||||||
|
|
||||||
|
func readRepositoryFile(t *testing.T, parts ...string) []byte {
|
||||||
|
t.Helper()
|
||||||
|
data, err := os.ReadFile(repositoryPath(parts...))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return data
|
||||||
|
}
|
||||||
323
internal/cli/recompute_execution_contract_test.go
Normal file
323
internal/cli/recompute_execution_contract_test.go
Normal file
@@ -0,0 +1,323 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"reflect"
|
||||||
|
"sort"
|
||||||
|
"strings"
|
||||||
|
"sync"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestRecomputeStepRecoversThroughFilesystemCheckpoints(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
invalidateOutput bool
|
||||||
|
wantCode int
|
||||||
|
}{
|
||||||
|
{name: "accepted producer is hydrated", wantCode: 0},
|
||||||
|
{name: "invalid producer stops dependents", invalidateOutput: true, wantCode: 1},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
roots := newRecomputeTestRoots(t)
|
||||||
|
harness := newRecomputeTestHarness()
|
||||||
|
fresh := runRecomputeCommand(roots, harness.options(), false)
|
||||||
|
if fresh.code != 0 {
|
||||||
|
t.Fatalf("fresh run code=%d stderr=%q", fresh.code, fresh.stderr)
|
||||||
|
}
|
||||||
|
removeCheckpointLaneStage(t, roots.checkpoints, "extract", "first", "producer")
|
||||||
|
removeCheckpointLaneStage(t, roots.checkpoints, "merge", "first", "producer")
|
||||||
|
if tt.invalidateOutput {
|
||||||
|
path := findCheckpointFile(t, roots.checkpoints, "normalize", "first", "producer", "output.json")
|
||||||
|
if err := os.WriteFile(path, []byte("{"), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
harness.resetCalls()
|
||||||
|
|
||||||
|
resumed := runRecomputeCommand(roots, harness.options(), true)
|
||||||
|
if resumed.code != tt.wantCode {
|
||||||
|
t.Fatalf("resumed code=%d stdout=%q stderr=%q", resumed.code, resumed.stdout, resumed.stderr)
|
||||||
|
}
|
||||||
|
events := readLatestCheckpointEvents(t, roots.debug)
|
||||||
|
if tt.invalidateOutput {
|
||||||
|
if harness.callsFor("test/extract/middle") != 0 || harness.callsFor("test/extract/dependent") != 0 {
|
||||||
|
t.Fatalf("dependent calls after invalid producer = %#v", harness.callsSnapshot())
|
||||||
|
}
|
||||||
|
if !strings.Contains(resumed.stderr, string(pipeline.CheckpointReasonDecodeFailed)) {
|
||||||
|
t.Fatalf("stderr=%q, want stable checkpoint reason", resumed.stderr)
|
||||||
|
}
|
||||||
|
assertNormalizeDecisionSequence(t, events, []checkpointDecisionExpectation{{"first", "producer", pipeline.CheckpointDecisionExecuted, pipeline.CheckpointReasonDecodeFailed}})
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
if got := harness.callsSnapshot(); !reflect.DeepEqual(got, map[string]int{"test/extract/dependent": 1, "test/extract/middle": 1}) {
|
||||||
|
t.Fatalf("resumed extractor calls = %#v", got)
|
||||||
|
}
|
||||||
|
outputPath := filepath.Join(latestChildDir(t, roots.output), "result.json")
|
||||||
|
data, err := os.ReadFile(outputPath)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if string(data) != "[\"producer\",\"unrelated\",\"middle\",\"dependent\"]\n" {
|
||||||
|
t.Fatalf("ordered output = %q", data)
|
||||||
|
}
|
||||||
|
assertNormalizeDecisionSequence(t, events, []checkpointDecisionExpectation{
|
||||||
|
{"first", "producer", pipeline.CheckpointDecisionReused, pipeline.CheckpointReasonAcceptedArtifactReused},
|
||||||
|
{"first", "unrelated", pipeline.CheckpointDecisionReused, pipeline.CheckpointReasonReused},
|
||||||
|
{"second", "middle", pipeline.CheckpointDecisionForcedRecompute, pipeline.CheckpointReasonRecomputeStep},
|
||||||
|
{"third", "dependent", pipeline.CheckpointDecisionForcedRecompute, pipeline.CheckpointReasonRecomputeStep},
|
||||||
|
})
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
type checkpointDecisionExpectation struct {
|
||||||
|
step, lane string
|
||||||
|
action pipeline.CheckpointDecisionCategory
|
||||||
|
reason pipeline.CheckpointReasonCode
|
||||||
|
}
|
||||||
|
|
||||||
|
func assertNormalizeDecisionSequence(t *testing.T, events []pipeline.CheckpointEvent, want []checkpointDecisionExpectation) {
|
||||||
|
t.Helper()
|
||||||
|
var got []checkpointDecisionExpectation
|
||||||
|
for _, event := range events {
|
||||||
|
if event.Stage == string(pipeline.StageNormalize) {
|
||||||
|
got = append(got, checkpointDecisionExpectation{event.StepID, event.LaneID, event.Action, event.ReasonCode})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !reflect.DeepEqual(got, want) {
|
||||||
|
t.Fatalf("normalize decisions = %#v, want %#v", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
type recomputeTestHarness struct {
|
||||||
|
base *stateTestHarness
|
||||||
|
mu sync.Mutex
|
||||||
|
calls map[string]int
|
||||||
|
}
|
||||||
|
|
||||||
|
func newRecomputeTestHarness() *recomputeTestHarness {
|
||||||
|
return &recomputeTestHarness{base: newStateTestHarness(), calls: make(map[string]int)}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (h *recomputeTestHarness) options() Options {
|
||||||
|
opts := h.base.options()
|
||||||
|
for _, key := range []string{"test/extract/producer", "test/extract/unrelated", "test/extract/middle", "test/extract/dependent"} {
|
||||||
|
moduleKey := key
|
||||||
|
spec := pipeline.ModuleSpec{
|
||||||
|
Key: moduleKey, Stage: pipeline.StageExtract, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: stateTestArtifactKind,
|
||||||
|
ReferenceSlots: []contracts.ReferenceSlot{{Name: "upstream", AcceptedMediaTypes: []string{"application/json"}, AcceptedArtifactKinds: []contracts.ArtifactKind{stateTestArtifactKind}}},
|
||||||
|
}
|
||||||
|
if err := pipeline.RegisterExtractor(opts.Registries.Extractors, spec, func() (contracts.Extractor[stateTestArtifact], error) {
|
||||||
|
return recomputeTestExtractor{key: moduleKey, harness: h}, nil
|
||||||
|
}); err != nil {
|
||||||
|
panic(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if err := opts.Registries.Outputs.RegisterWithSpec(pipeline.ModuleSpec{Key: "test/recompute-output", Stage: pipeline.StageOutput, Requires: []string{"normalized"}, Provides: []string{"output"}}, func() (contracts.OutputEncoder, error) {
|
||||||
|
return recomputeTestOutput{}, nil
|
||||||
|
}); err != nil {
|
||||||
|
panic(err)
|
||||||
|
}
|
||||||
|
opts.Catalog = catalogFromRegistries(opts.Registries)
|
||||||
|
return opts
|
||||||
|
}
|
||||||
|
|
||||||
|
func (h *recomputeTestHarness) record(key string) {
|
||||||
|
h.mu.Lock()
|
||||||
|
defer h.mu.Unlock()
|
||||||
|
h.calls[key]++
|
||||||
|
}
|
||||||
|
|
||||||
|
func (h *recomputeTestHarness) resetCalls() {
|
||||||
|
h.mu.Lock()
|
||||||
|
defer h.mu.Unlock()
|
||||||
|
h.calls = make(map[string]int)
|
||||||
|
}
|
||||||
|
|
||||||
|
func (h *recomputeTestHarness) callsFor(key string) int {
|
||||||
|
h.mu.Lock()
|
||||||
|
defer h.mu.Unlock()
|
||||||
|
return h.calls[key]
|
||||||
|
}
|
||||||
|
|
||||||
|
func (h *recomputeTestHarness) callsSnapshot() map[string]int {
|
||||||
|
h.mu.Lock()
|
||||||
|
defer h.mu.Unlock()
|
||||||
|
result := make(map[string]int, len(h.calls))
|
||||||
|
for key, value := range h.calls {
|
||||||
|
result[key] = value
|
||||||
|
}
|
||||||
|
return result
|
||||||
|
}
|
||||||
|
|
||||||
|
type recomputeTestExtractor struct {
|
||||||
|
key string
|
||||||
|
harness *recomputeTestHarness
|
||||||
|
}
|
||||||
|
|
||||||
|
func (e recomputeTestExtractor) Key() string { return e.key }
|
||||||
|
func (e recomputeTestExtractor) ReferenceSlots() []contracts.ReferenceSlot {
|
||||||
|
return []contracts.ReferenceSlot{{Name: "upstream", AcceptedMediaTypes: []string{"application/json"}, AcceptedArtifactKinds: []contracts.ArtifactKind{stateTestArtifactKind}}}
|
||||||
|
}
|
||||||
|
func (e recomputeTestExtractor) Extract(context.Context, contracts.TypedExtractionRequest) (contracts.TypedExtractionResult[stateTestArtifact], error) {
|
||||||
|
e.harness.record(e.key)
|
||||||
|
return contracts.TypedExtractionResult[stateTestArtifact]{Value: stateTestArtifact{Value: e.key}}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type recomputeTestOutput struct{}
|
||||||
|
|
||||||
|
func (recomputeTestOutput) Key() string { return "test/recompute-output" }
|
||||||
|
func (recomputeTestOutput) Encode(_ context.Context, req contracts.OutputRequest) (contracts.OutputResult, error) {
|
||||||
|
lanes := make([]string, 0, len(req.NormalizeOutputs))
|
||||||
|
for _, output := range req.NormalizeOutputs {
|
||||||
|
lanes = append(lanes, output.LaneID)
|
||||||
|
}
|
||||||
|
data, err := json.Marshal(lanes)
|
||||||
|
if err != nil {
|
||||||
|
return contracts.OutputResult{}, err
|
||||||
|
}
|
||||||
|
return contracts.OutputResult{Files: []contracts.OutputFile{{Name: "result.json", Bytes: append(data, '\n')}}}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func newRecomputeTestRoots(t *testing.T) stateTestRoots {
|
||||||
|
t.Helper()
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
config := fmt.Sprintf(`version: 3
|
||||||
|
output:
|
||||||
|
directory: %q
|
||||||
|
cache:
|
||||||
|
chunk_plans:
|
||||||
|
directory: %q
|
||||||
|
mode: bypass
|
||||||
|
checkpoints:
|
||||||
|
enabled: true
|
||||||
|
directory: %q
|
||||||
|
debug:
|
||||||
|
directory: %q
|
||||||
|
pipelines:
|
||||||
|
sample:
|
||||||
|
input: test/input
|
||||||
|
chunk: test/chunk
|
||||||
|
steps:
|
||||||
|
- id: first
|
||||||
|
artifacts:
|
||||||
|
producer:
|
||||||
|
extract: test/extract/producer
|
||||||
|
merge: test/merge
|
||||||
|
normalize: test/normalize
|
||||||
|
unrelated:
|
||||||
|
extract: test/extract/unrelated
|
||||||
|
merge: test/merge
|
||||||
|
normalize: test/normalize
|
||||||
|
- id: second
|
||||||
|
references:
|
||||||
|
upstream:
|
||||||
|
artifact:
|
||||||
|
step: first
|
||||||
|
lane: producer
|
||||||
|
artifacts:
|
||||||
|
middle:
|
||||||
|
extract: test/extract/middle
|
||||||
|
merge: test/merge
|
||||||
|
normalize: test/normalize
|
||||||
|
- id: third
|
||||||
|
references:
|
||||||
|
upstream:
|
||||||
|
artifact:
|
||||||
|
step: second
|
||||||
|
lane: middle
|
||||||
|
artifacts:
|
||||||
|
dependent:
|
||||||
|
extract: test/extract/dependent
|
||||||
|
merge: test/merge
|
||||||
|
normalize: test/normalize
|
||||||
|
output: test/recompute-output
|
||||||
|
`, roots.output, roots.plans, roots.checkpoints, roots.debug)
|
||||||
|
if err := os.WriteFile(roots.config, []byte(config), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return roots
|
||||||
|
}
|
||||||
|
|
||||||
|
func runRecomputeCommand(roots stateTestRoots, opts Options, recompute bool) stateTestResult {
|
||||||
|
args := []string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass"}
|
||||||
|
if recompute {
|
||||||
|
args = append(args, "--resume", "--recompute-step", "second", "--debug")
|
||||||
|
}
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
return stateTestResult{code: RunWithOptions(args, &stdout, &stderr, opts), stdout: stdout.String(), stderr: stderr.String()}
|
||||||
|
}
|
||||||
|
|
||||||
|
func removeCheckpointLaneStage(t *testing.T, root, stage, step, lane string) {
|
||||||
|
t.Helper()
|
||||||
|
dir := filepath.Dir(findCheckpointFile(t, root, stage, step, lane, "manifest.json"))
|
||||||
|
if err := os.RemoveAll(dir); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func findCheckpointFile(t *testing.T, root, stage, step, lane, name string) string {
|
||||||
|
t.Helper()
|
||||||
|
want := filepath.Join(stage, step, lane, name)
|
||||||
|
var matches []string
|
||||||
|
err := filepath.WalkDir(root, func(path string, entry os.DirEntry, err error) error {
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if !entry.IsDir() && strings.HasSuffix(path, want) {
|
||||||
|
matches = append(matches, path)
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if len(matches) != 1 {
|
||||||
|
t.Fatalf("checkpoint files ending in %q = %v", want, matches)
|
||||||
|
}
|
||||||
|
return matches[0]
|
||||||
|
}
|
||||||
|
|
||||||
|
func latestChildDir(t *testing.T, root string) string {
|
||||||
|
t.Helper()
|
||||||
|
entries, err := os.ReadDir(root)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var dirs []string
|
||||||
|
for _, entry := range entries {
|
||||||
|
if entry.IsDir() {
|
||||||
|
dirs = append(dirs, filepath.Join(root, entry.Name()))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(dirs) == 0 {
|
||||||
|
t.Fatal("no child directory")
|
||||||
|
}
|
||||||
|
sort.Strings(dirs)
|
||||||
|
return dirs[len(dirs)-1]
|
||||||
|
}
|
||||||
|
|
||||||
|
func readLatestCheckpointEvents(t *testing.T, root string) []pipeline.CheckpointEvent {
|
||||||
|
t.Helper()
|
||||||
|
var events []pipeline.CheckpointEvent
|
||||||
|
data, err := os.ReadFile(filepath.Join(latestChildDir(t, root), "summary", "checkpoint-events.json"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := json.Unmarshal(data, &events); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return events
|
||||||
|
}
|
||||||
56
internal/cli/recompute_policy_test.go
Normal file
56
internal/cli/recompute_policy_test.go
Normal file
@@ -0,0 +1,56 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestRecomputePolicyIncludesDependentLanesAndReusablePredecessors(t *testing.T) {
|
||||||
|
producer := pipeline.ResolvedArtifactLane{StepID: "first", ID: "producer"}
|
||||||
|
unrelated := pipeline.ResolvedArtifactLane{StepID: "first", ID: "unrelated"}
|
||||||
|
consumer := pipeline.ResolvedArtifactLane{
|
||||||
|
StepID: "second", ID: "consumer",
|
||||||
|
ExtractReferences: pipeline.ResolvedReferenceTarget{Bindings: []pipeline.ReferenceBinding{{Artifact: &pipeline.ArtifactReference{Step: "first", Lane: "producer"}}}},
|
||||||
|
}
|
||||||
|
downstream := pipeline.ResolvedArtifactLane{
|
||||||
|
StepID: "third", ID: "downstream",
|
||||||
|
ExtractReferences: pipeline.ResolvedReferenceTarget{Bindings: []pipeline.ReferenceBinding{{Artifact: &pipeline.ArtifactReference{Step: "second", Lane: "consumer"}}}},
|
||||||
|
}
|
||||||
|
independent := pipeline.ResolvedArtifactLane{StepID: "third", ID: "independent"}
|
||||||
|
resolved := pipeline.ResolvedPipeline{Steps: []pipeline.ResolvedPipelineStep{
|
||||||
|
{ID: "first", ArtifactLanes: []pipeline.ResolvedArtifactLane{producer, unrelated}},
|
||||||
|
{ID: "second", ArtifactLanes: []pipeline.ResolvedArtifactLane{consumer}},
|
||||||
|
{ID: "third", ArtifactLanes: []pipeline.ResolvedArtifactLane{downstream, independent}},
|
||||||
|
}}
|
||||||
|
|
||||||
|
policy, err := recomputePolicy(resolved, "second")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if _, ok := policy.ForcedLanes[pipeline.CheckpointLaneKey("second", "consumer")]; !ok {
|
||||||
|
t.Fatal("selected lane was not forced")
|
||||||
|
}
|
||||||
|
if _, ok := policy.ForcedLanes[pipeline.CheckpointLaneKey("third", "downstream")]; !ok {
|
||||||
|
t.Fatal("transitive dependent lane was not forced")
|
||||||
|
}
|
||||||
|
if _, ok := policy.ForcedLanes[pipeline.CheckpointLaneKey("first", "producer")]; ok {
|
||||||
|
t.Fatal("predecessor was implicitly forced")
|
||||||
|
}
|
||||||
|
if _, ok := policy.RequireReusableLanes[pipeline.CheckpointLaneKey("first", "producer")]; !ok {
|
||||||
|
t.Fatal("required predecessor was not marked reusable")
|
||||||
|
}
|
||||||
|
if _, ok := policy.ForcedLanes[pipeline.CheckpointLaneKey("first", "unrelated")]; ok {
|
||||||
|
t.Fatal("unrelated lane was forced")
|
||||||
|
}
|
||||||
|
if _, ok := policy.ForcedLanes[pipeline.CheckpointLaneKey("third", "independent")]; ok {
|
||||||
|
t.Fatal("unrelated later lane was forced")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRecomputePolicyRejectsUnknownStep(t *testing.T) {
|
||||||
|
_, err := recomputePolicy(pipeline.ResolvedPipeline{Steps: []pipeline.ResolvedPipelineStep{{ID: "known"}}}, "missing")
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("unknown step was accepted")
|
||||||
|
}
|
||||||
|
}
|
||||||
456
internal/cli/reference_contract_test.go
Normal file
456
internal/cli/reference_contract_test.go
Normal file
@@ -0,0 +1,456 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestReferenceSelectorsParseAndApplyAllDocumentedForms(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
selector string
|
||||||
|
only []string
|
||||||
|
wantStage pipeline.ModuleStage
|
||||||
|
wantLane string
|
||||||
|
wantSlot string
|
||||||
|
}{
|
||||||
|
{name: "flat", selector: "alpha-slot", wantStage: pipeline.StageExtract, wantLane: "alpha", wantSlot: "alpha-slot"},
|
||||||
|
{name: "chunk", selector: "chunk.chunk-slot", wantStage: pipeline.StageChunk, wantSlot: "chunk-slot"},
|
||||||
|
{name: "merge", selector: "merge.alpha-merge", only: []string{"alpha"}, wantStage: pipeline.StageMerge, wantLane: "alpha", wantSlot: "alpha-merge"},
|
||||||
|
{name: "lane", selector: "alpha.alpha-slot", wantStage: pipeline.StageExtract, wantLane: "alpha", wantSlot: "alpha-slot"},
|
||||||
|
{name: "lane extract", selector: "alpha.extract.alpha-slot", wantStage: pipeline.StageExtract, wantLane: "alpha", wantSlot: "alpha-slot"},
|
||||||
|
{name: "lane merge", selector: "alpha.merge.alpha-merge", wantStage: pipeline.StageMerge, wantLane: "alpha", wantSlot: "alpha-merge"},
|
||||||
|
{name: "lane normalize", selector: "alpha.normalize.alpha-normalize", wantStage: pipeline.StageNormalize, wantLane: "alpha", wantSlot: "alpha-normalize"},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
cfg := referenceContractConfig()
|
||||||
|
catalog := referenceContractCatalog(t, true, true)
|
||||||
|
selector, err := parseReferenceSelector(tt.selector, "--reference")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
overrides, _, err := resolveCLIReferenceRequests(cfg, "demo", tt.only, catalog, []cliReferenceRequest{{Selector: selector, Source: "reference.txt"}}, nil)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("resolve selector: %v", err)
|
||||||
|
}
|
||||||
|
if len(overrides) != 1 {
|
||||||
|
t.Fatalf("overrides = %#v, want one binding", overrides)
|
||||||
|
}
|
||||||
|
got := overrides[0]
|
||||||
|
if got.Stage != tt.wantStage || got.LaneID != tt.wantLane || got.SlotName != tt.wantSlot || got.BindingSource != contracts.ReferenceBindingSourceCLI {
|
||||||
|
t.Fatalf("binding = %#v, want %s/%s/%s from CLI", got, tt.wantStage, tt.wantLane, tt.wantSlot)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestReferenceSelectorsRejectAmbiguityWithSpecificSuggestions(t *testing.T) {
|
||||||
|
cfg := referenceContractConfig()
|
||||||
|
catalog := referenceContractCatalog(t, true, true)
|
||||||
|
for _, tt := range []struct {
|
||||||
|
name string
|
||||||
|
selector string
|
||||||
|
want []string
|
||||||
|
}{
|
||||||
|
{name: "flat shared slot", selector: "shared", want: []string{"alpha.extract.shared", "beta.extract.shared"}},
|
||||||
|
{name: "lane shared slot", selector: "alpha.shared", want: []string{"alpha.extract.shared", "alpha.merge.shared", "alpha.normalize.shared"}},
|
||||||
|
{name: "all mergers", selector: "merge.shared", want: []string{"alpha.merge.shared", "beta.merge.shared"}},
|
||||||
|
} {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
selector, err := parseReferenceSelector(tt.selector, "--reference")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
_, _, err = resolveCLIReferenceRequests(cfg, "demo", nil, catalog, []cliReferenceRequest{{Selector: selector, Source: "reference.txt"}}, nil)
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("resolve selector succeeded, want ambiguity error")
|
||||||
|
}
|
||||||
|
for _, fragment := range tt.want {
|
||||||
|
if !strings.Contains(err.Error(), fragment) {
|
||||||
|
t.Fatalf("error = %q, want suggestion %q", err, fragment)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestReferenceSelectorsRespectSelectedLanesBeforeMaterialization(t *testing.T) {
|
||||||
|
cfg := referenceContractConfig()
|
||||||
|
catalog := referenceContractCatalog(t, true, true)
|
||||||
|
for _, tt := range []struct {
|
||||||
|
name string
|
||||||
|
selector string
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{name: "unselected lane", selector: "beta.extract.beta-slot", want: `reference lane "beta" is not selected`},
|
||||||
|
{name: "unknown lane", selector: "missing.extract.beta-slot", want: `reference lane "missing" is not selected`},
|
||||||
|
} {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
selector, err := parseReferenceSelector(tt.selector, "--reference")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
_, _, err = resolveCLIReferenceRequests(cfg, "demo", []string{"alpha"}, catalog, []cliReferenceRequest{{Selector: selector, Source: filepath.Join(t.TempDir(), "missing.txt")}}, nil)
|
||||||
|
if err == nil || !strings.Contains(err.Error(), tt.want) || strings.Contains(err.Error(), "missing.txt") {
|
||||||
|
t.Fatalf("error = %v, want selection failure before file access", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestReferenceSyntaxErrorsReturnTwo(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
args []string
|
||||||
|
}{
|
||||||
|
{name: "reference missing value", args: []string{"run", "demo", "--config", "missing.yml", "--input", "input.txt", "--reference"}},
|
||||||
|
{name: "reference missing selector", args: []string{"run", "demo", "--config", "missing.yml", "--input", "input.txt", "--reference", "=path.txt"}},
|
||||||
|
{name: "reference missing separator", args: []string{"run", "demo", "--config", "missing.yml", "--input", "input.txt", "--reference", "slot"}},
|
||||||
|
{name: "reference missing path", args: []string{"run", "demo", "--config", "missing.yml", "--input", "input.txt", "--reference", "slot="}},
|
||||||
|
{name: "reference excess segments", args: []string{"run", "demo", "--config", "missing.yml", "--input", "input.txt", "--reference", "a.b.c.d=path.txt"}},
|
||||||
|
{name: "unbind with path", args: []string{"run", "demo", "--config", "missing.yml", "--input", "input.txt", "--without-reference", "slot=path.txt"}},
|
||||||
|
{name: "unbind excess segments", args: []string{"run", "demo", "--config", "missing.yml", "--input", "input.txt", "--without-reference", "a.b.c.d"}},
|
||||||
|
{name: "unbind missing value", args: []string{"run", "demo", "--config", "missing.yml", "--input", "input.txt", "--without-reference"}},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions(tt.args, &stdout, &stderr, Options{LookupEnv: emptyLookup})
|
||||||
|
if code != 2 || stdout.Len() != 0 || stderr.Len() == 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestReferenceOverridesUseFinalExactTargetBinding(t *testing.T) {
|
||||||
|
cfg := referenceContractConfig()
|
||||||
|
catalog := referenceContractCatalog(t, true, true)
|
||||||
|
alphaShared, err := parseReferenceSelector("alpha.extract.shared", "--reference")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
betaShared, err := parseReferenceSelector("beta.extract.shared", "--reference")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
overrides, unbinds, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog, []cliReferenceRequest{
|
||||||
|
{Selector: alphaShared, Source: "alpha-first.txt"},
|
||||||
|
{Selector: alphaShared, Source: "alpha-final.txt"},
|
||||||
|
{Selector: betaShared, Source: "beta-only.txt"},
|
||||||
|
}, nil)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if len(unbinds) != 0 {
|
||||||
|
t.Fatalf("unbinds = %#v, want none", unbinds)
|
||||||
|
}
|
||||||
|
effective, err := cfg.Resolve(config.ResolveInput{PipelineID: "demo", Catalog: catalog, ReferenceOverrides: overrides})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("resolve pipeline: %v", err)
|
||||||
|
}
|
||||||
|
alpha := referenceContractLane(t, effective.ResolvedPipeline, "alpha")
|
||||||
|
beta := referenceContractLane(t, effective.ResolvedPipeline, "beta")
|
||||||
|
if source := referenceContractBindingSource(alpha.ExtractReferences.Bindings, "shared"); source != "alpha-final.txt" {
|
||||||
|
t.Fatalf("alpha shared source = %q, want final exact-target override", source)
|
||||||
|
}
|
||||||
|
if source := referenceContractBindingSource(beta.ExtractReferences.Bindings, "shared"); source != "beta-only.txt" {
|
||||||
|
t.Fatalf("beta shared source = %q, want target-specific override", source)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestReferenceUnbindsRemoveOptionalAndProtectRequiredSlots(t *testing.T) {
|
||||||
|
cfg := referenceContractConfig()
|
||||||
|
catalog := referenceContractCatalog(t, true, true)
|
||||||
|
optional, err := parseReferenceSelector("alpha.extract.alpha-slot", "--without-reference")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
_, without, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog, nil, []cliReferenceUnbindRequest{{Selector: optional}})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
effective, err := cfg.Resolve(config.ResolveInput{PipelineID: "demo", Catalog: catalog, ReferenceUnbinds: without})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("optional unbind: %v", err)
|
||||||
|
}
|
||||||
|
if binding := referenceContractFindBinding(referenceContractLane(t, effective.ResolvedPipeline, "alpha").ExtractReferences.Bindings, "alpha-slot"); binding != nil {
|
||||||
|
t.Fatalf("optional binding after unbind = %#v, want absent", binding)
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range []struct {
|
||||||
|
name string
|
||||||
|
selector string
|
||||||
|
}{
|
||||||
|
{name: "chunk", selector: "chunk.required-chunk"},
|
||||||
|
{name: "extract", selector: "alpha.extract.required-extract"},
|
||||||
|
{name: "merge", selector: "alpha.merge.required-merge"},
|
||||||
|
{name: "normalize", selector: "alpha.normalize.required-normalize"},
|
||||||
|
} {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
selector, err := parseReferenceSelector(tt.selector, "--without-reference")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
_, unbinds, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog, nil, []cliReferenceUnbindRequest{{Selector: selector}})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
_, err = cfg.Resolve(config.ResolveInput{PipelineID: "demo", Catalog: catalog, ReferenceUnbinds: unbinds})
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "required reference slot") {
|
||||||
|
t.Fatalf("resolve error = %v, want required-slot failure", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestReferenceMaterializationSeparatesCLIAndConfigPathOrigins(t *testing.T) {
|
||||||
|
configDir := t.TempDir()
|
||||||
|
workingDir := t.TempDir()
|
||||||
|
cfg := referenceContractConfig()
|
||||||
|
configPath := filepath.Join(configDir, "config.yml")
|
||||||
|
if err := os.WriteFile(configPath, []byte("version: 3\n"), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(filepath.Join(configDir, "required.txt"), []byte("config reference"), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(filepath.Join(configDir, "optional.txt"), []byte("optional reference"), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(filepath.Join(workingDir, "cli-reference.txt"), []byte("CLI reference"), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
catalog := referenceContractCatalog(t, true, true)
|
||||||
|
selector, err := parseReferenceSelector("alpha.extract.alpha-slot", "--reference")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
overrides, unbinds, err := resolveCLIReferenceRequests(cfg, "demo", nil, catalog, []cliReferenceRequest{{Selector: selector, Source: "cli-reference.txt"}}, nil)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
effective, err := cfg.Resolve(config.ResolveInput{PipelineID: "demo", Catalog: catalog, ReferenceOverrides: overrides, ReferenceUnbinds: unbinds})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("resolve pipeline: %v", err)
|
||||||
|
}
|
||||||
|
materialized, _, err := pipeline.MaterializeReferences(effective.ResolvedPipeline, catalog, pipeline.ReferenceMaterializationOptions{ConfigPath: configPath, WorkingDir: workingDir})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("materialize references: %v", err)
|
||||||
|
}
|
||||||
|
alpha := referenceContractLane(t, materialized, "alpha")
|
||||||
|
cliItem := alpha.ExtractReferences.ReferenceSet.Slots["alpha-slot"].Items[0]
|
||||||
|
if string(cliItem.Content) != "CLI reference" || cliItem.BindingSource != contracts.ReferenceBindingSourceCLI || cliItem.Origin.URI != referenceContractFileURI(filepath.Join(workingDir, "cli-reference.txt")) {
|
||||||
|
t.Fatalf("CLI materialization = %#v, want working-directory provenance", cliItem)
|
||||||
|
}
|
||||||
|
configItem := alpha.ExtractReferences.ReferenceSet.Slots["required-extract"].Items[0]
|
||||||
|
if string(configItem.Content) != "config reference" || configItem.BindingSource != contracts.ReferenceBindingSourceConfig || configItem.Origin.URI != referenceContractFileURI(filepath.Join(configDir, "required.txt")) {
|
||||||
|
t.Fatalf("config materialization = %#v, want config-directory provenance", configItem)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestReferenceTargetLookupUsesArtifactVariantsAndReportsMissingContext(t *testing.T) {
|
||||||
|
cfg := referenceContractConfig()
|
||||||
|
full := referenceContractCatalog(t, true, true)
|
||||||
|
targets, err := selectedReferenceTargets(cfg, "demo", nil, full)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("select reference targets: %v", err)
|
||||||
|
}
|
||||||
|
var alphaMerge, betaMerge selectedReferenceTarget
|
||||||
|
for _, target := range targets {
|
||||||
|
if target.stage == pipeline.StageMerge && target.laneID == "alpha" {
|
||||||
|
alphaMerge = target
|
||||||
|
}
|
||||||
|
if target.stage == pipeline.StageMerge && target.laneID == "beta" {
|
||||||
|
betaMerge = target
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if _, ok := alphaMerge.slots["alpha-merge"]; !ok {
|
||||||
|
t.Fatalf("alpha merger slots = %#v, want alpha artifact variant", alphaMerge.slots)
|
||||||
|
}
|
||||||
|
if _, ok := betaMerge.slots["beta-merge"]; !ok {
|
||||||
|
t.Fatalf("beta merger slots = %#v, want beta artifact variant", betaMerge.slots)
|
||||||
|
}
|
||||||
|
if _, ok := betaMerge.slots["alpha-merge"]; ok {
|
||||||
|
t.Fatalf("beta merger slots = %#v, must not use alpha variant", betaMerge.slots)
|
||||||
|
}
|
||||||
|
|
||||||
|
missingMerger := referenceContractCatalog(t, false, true)
|
||||||
|
_, err = selectedReferenceTargets(cfg, "demo", nil, missingMerger)
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "merger") || !strings.Contains(err.Error(), string(referenceContractKindBeta)) {
|
||||||
|
t.Fatalf("missing merger error = %v, want artifact variant context", err)
|
||||||
|
}
|
||||||
|
missingNormalizer := referenceContractCatalog(t, true, false)
|
||||||
|
_, err = selectedReferenceTargets(cfg, "demo", nil, missingNormalizer)
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "normalizer") || !strings.Contains(err.Error(), string(referenceContractKindBeta)) {
|
||||||
|
t.Fatalf("missing normalizer error = %v, want artifact variant context", err)
|
||||||
|
}
|
||||||
|
missingExtractor := referenceContractCatalog(t, true, true)
|
||||||
|
missingExtractor.Extractors = pipeline.NewExtractorRegistry()
|
||||||
|
_, err = selectedReferenceTargets(cfg, "demo", nil, missingExtractor)
|
||||||
|
if err == nil || !strings.Contains(err.Error(), `lane "alpha" extract module`) || !strings.Contains(err.Error(), "not registered") {
|
||||||
|
t.Fatalf("missing extractor error = %v, want lane/module context", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const (
|
||||||
|
referenceContractKindAlpha contracts.ArtifactKind = "reference/alpha"
|
||||||
|
referenceContractKindBeta contracts.ArtifactKind = "reference/beta"
|
||||||
|
)
|
||||||
|
|
||||||
|
func referenceContractConfig() config.Config {
|
||||||
|
cfg := config.Default()
|
||||||
|
cfg.Pipelines = map[string]pipeline.PipelineProfile{
|
||||||
|
"demo": {
|
||||||
|
ID: "demo",
|
||||||
|
Input: pipeline.Binding("reference/input"),
|
||||||
|
Chunk: pipeline.Binding("reference/chunk"),
|
||||||
|
Output: pipeline.Binding("reference/output"),
|
||||||
|
Artifacts: map[string]pipeline.ArtifactLaneProfile{
|
||||||
|
"alpha": {
|
||||||
|
Extract: pipeline.Binding("reference/extract-alpha"),
|
||||||
|
Merge: pipeline.Binding("reference/shared-merge"),
|
||||||
|
Normalize: pipeline.Binding("reference/shared-normalize"),
|
||||||
|
References: pipeline.ExternalReferenceMap(map[string]string{"required-extract": "required.txt"}),
|
||||||
|
},
|
||||||
|
"beta": {
|
||||||
|
Extract: pipeline.Binding("reference/extract-beta"),
|
||||||
|
Merge: pipeline.Binding("reference/shared-merge"),
|
||||||
|
Normalize: pipeline.Binding("reference/shared-normalize"),
|
||||||
|
References: pipeline.ExternalReferenceMap(map[string]string{"required-extract": "required.txt"}),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
profile := cfg.Pipelines["demo"]
|
||||||
|
profile.Chunk.References = pipeline.ExternalReferenceMap(map[string]string{"required-chunk": "required.txt"})
|
||||||
|
alpha := profile.Artifacts["alpha"]
|
||||||
|
alpha.Extract.References = pipeline.ExternalReferenceMap(map[string]string{"required-extract": "required.txt", "alpha-slot": "optional.txt"})
|
||||||
|
alpha.Merge.References = pipeline.ExternalReferenceMap(map[string]string{"required-merge": "required.txt"})
|
||||||
|
alpha.Normalize.References = pipeline.ExternalReferenceMap(map[string]string{"required-normalize": "required.txt"})
|
||||||
|
profile.Artifacts["alpha"] = alpha
|
||||||
|
beta := profile.Artifacts["beta"]
|
||||||
|
beta.Extract.References = pipeline.ExternalReferenceMap(map[string]string{"required-extract": "required.txt"})
|
||||||
|
beta.Merge.References = pipeline.ExternalReferenceMap(map[string]string{"required-merge": "required.txt"})
|
||||||
|
beta.Normalize.References = pipeline.ExternalReferenceMap(map[string]string{"required-normalize": "required.txt"})
|
||||||
|
profile.Artifacts["beta"] = beta
|
||||||
|
cfg.Pipelines["demo"] = profile
|
||||||
|
return cfg
|
||||||
|
}
|
||||||
|
|
||||||
|
func referenceContractCatalog(t *testing.T, includeBetaMerger, includeBetaNormalizer bool) pipeline.ModuleCatalog {
|
||||||
|
t.Helper()
|
||||||
|
registries := pipeline.Registries{
|
||||||
|
Inputs: pipeline.NewInputAdapterRegistry(),
|
||||||
|
Chunkers: pipeline.NewChunkerRegistry(),
|
||||||
|
ArtifactCodecs: pipeline.NewArtifactCodecRegistry(),
|
||||||
|
Extractors: pipeline.NewExtractorRegistry(),
|
||||||
|
Mergers: pipeline.NewMergerRegistry(),
|
||||||
|
Normalizers: pipeline.NewNormalizerRegistry(),
|
||||||
|
Validators: pipeline.NewValidatorRegistry(),
|
||||||
|
ValidatorChains: pipeline.NewValidatorChainRegistry(),
|
||||||
|
Outputs: pipeline.NewOutputEncoderRegistry(),
|
||||||
|
}
|
||||||
|
register := func(err error) {
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
register(registries.Inputs.RegisterBuilderWithSpec(pipeline.ModuleSpec{Key: "reference/input", Stage: pipeline.StageInput, Provides: []string{"source"}}, func(map[string]any) error { return nil }, func(pipeline.BuildRequest) (contracts.InputAdapter, error) { return stateTestInput{}, nil }))
|
||||||
|
register(registries.Chunkers.RegisterWithSpec(pipeline.ModuleSpec{Key: "reference/chunk", Stage: pipeline.StageChunk, Requires: []string{"source"}, Provides: []string{"chunks"}, ReferenceSlots: []contracts.ReferenceSlot{{Name: "chunk-slot"}, {Name: "required-chunk", Required: true}}}, func() (contracts.Chunker, error) { return stateTestChunker{}, nil }))
|
||||||
|
register(pipeline.RegisterArtifactCodec(registries.ArtifactCodecs, referenceContractCodecA{}))
|
||||||
|
register(pipeline.RegisterArtifactCodec(registries.ArtifactCodecs, referenceContractCodecB{}))
|
||||||
|
register(pipeline.RegisterExtractor(registries.Extractors, pipeline.ModuleSpec{Key: "reference/extract-alpha", Stage: pipeline.StageExtract, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: referenceContractKindAlpha, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "alpha-slot"}, {Name: "required-extract", Required: true}}}, func() (contracts.Extractor[stateTestArtifact], error) { return stateTestExtractor{}, nil }))
|
||||||
|
register(pipeline.RegisterExtractor(registries.Extractors, pipeline.ModuleSpec{Key: "reference/extract-beta", Stage: pipeline.StageExtract, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: referenceContractKindBeta, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "beta-slot"}, {Name: "required-extract", Required: true}}}, func() (contracts.Extractor[stateTestArtifact], error) { return stateTestExtractor{}, nil }))
|
||||||
|
register(pipeline.RegisterMerger(registries.Mergers, pipeline.ModuleSpec{Key: "reference/shared-merge", Stage: pipeline.StageMerge, Requires: []string{"artifact"}, Provides: []string{"merged"}, ArtifactKind: referenceContractKindAlpha, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "alpha-merge"}, {Name: "required-merge", Required: true}}}, func() (contracts.Merger[stateTestArtifact], error) { return stateTestMerger{}, nil }))
|
||||||
|
if includeBetaMerger {
|
||||||
|
register(pipeline.RegisterMerger(registries.Mergers, pipeline.ModuleSpec{Key: "reference/shared-merge", Stage: pipeline.StageMerge, Requires: []string{"artifact"}, Provides: []string{"merged"}, ArtifactKind: referenceContractKindBeta, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "beta-merge"}, {Name: "required-merge", Required: true}}}, func() (contracts.Merger[stateTestArtifact], error) { return stateTestMerger{}, nil }))
|
||||||
|
}
|
||||||
|
register(pipeline.RegisterNormalizer(registries.Normalizers, pipeline.ModuleSpec{Key: "reference/shared-normalize", Stage: pipeline.StageNormalize, Requires: []string{"merged"}, Provides: []string{"normalized"}, ArtifactKind: referenceContractKindAlpha, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "alpha-normalize"}, {Name: "required-normalize", Required: true}}}, func() (contracts.Normalizer[stateTestArtifact], error) { return stateTestNormalizer{}, nil }))
|
||||||
|
if includeBetaNormalizer {
|
||||||
|
register(pipeline.RegisterNormalizer(registries.Normalizers, pipeline.ModuleSpec{Key: "reference/shared-normalize", Stage: pipeline.StageNormalize, Requires: []string{"merged"}, Provides: []string{"normalized"}, ArtifactKind: referenceContractKindBeta, ReferenceSlots: []contracts.ReferenceSlot{{Name: "shared"}, {Name: "beta-normalize"}, {Name: "required-normalize", Required: true}}}, func() (contracts.Normalizer[stateTestArtifact], error) { return stateTestNormalizer{}, nil }))
|
||||||
|
}
|
||||||
|
register(registries.Outputs.RegisterWithSpec(pipeline.ModuleSpec{Key: "reference/output", Stage: pipeline.StageOutput, Requires: []string{"normalized"}, Provides: []string{"output"}}, func() (contracts.OutputEncoder, error) { return stateTestOutput{}, nil }))
|
||||||
|
return catalogFromRegistries(registries)
|
||||||
|
}
|
||||||
|
|
||||||
|
type referenceContractCodecB struct{}
|
||||||
|
|
||||||
|
type referenceContractCodecA struct{}
|
||||||
|
|
||||||
|
func (referenceContractCodecA) Kind() contracts.ArtifactKind { return referenceContractKindAlpha }
|
||||||
|
func (referenceContractCodecA) Schema() contracts.ArtifactSchema {
|
||||||
|
return contracts.ArtifactSchema{ID: "reference.alpha", Name: "reference_alpha", Version: "v1", JSONSchema: []byte(`{"type":"object"}`)}
|
||||||
|
}
|
||||||
|
func (referenceContractCodecA) MediaType() string { return "application/json" }
|
||||||
|
func (referenceContractCodecA) EncodeCandidate(stateTestArtifact) ([]byte, error) {
|
||||||
|
return []byte(`{"value":"ok"}`), nil
|
||||||
|
}
|
||||||
|
func (referenceContractCodecA) Encode(stateTestArtifact) ([]byte, error) {
|
||||||
|
return []byte(`{"value":"ok"}`), nil
|
||||||
|
}
|
||||||
|
func (referenceContractCodecA) Decode([]byte) (stateTestArtifact, error) {
|
||||||
|
return stateTestArtifact{Value: "ok"}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (referenceContractCodecB) Kind() contracts.ArtifactKind { return referenceContractKindBeta }
|
||||||
|
func (referenceContractCodecB) Schema() contracts.ArtifactSchema {
|
||||||
|
return contracts.ArtifactSchema{ID: "reference.beta", Name: "reference_beta", Version: "v1", JSONSchema: []byte(`{"type":"object"}`)}
|
||||||
|
}
|
||||||
|
func (referenceContractCodecB) MediaType() string { return "application/json" }
|
||||||
|
func (referenceContractCodecB) EncodeCandidate(stateTestArtifact) ([]byte, error) {
|
||||||
|
return []byte(`{"value":"ok"}`), nil
|
||||||
|
}
|
||||||
|
func (referenceContractCodecB) Encode(stateTestArtifact) ([]byte, error) {
|
||||||
|
return []byte(`{"value":"ok"}`), nil
|
||||||
|
}
|
||||||
|
func (referenceContractCodecB) Decode([]byte) (stateTestArtifact, error) {
|
||||||
|
return stateTestArtifact{Value: "ok"}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func referenceContractLane(t *testing.T, resolved pipeline.ResolvedPipeline, id string) pipeline.ResolvedArtifactLane {
|
||||||
|
t.Helper()
|
||||||
|
for _, step := range resolved.Steps {
|
||||||
|
for _, lane := range step.ArtifactLanes {
|
||||||
|
if lane.ID == id {
|
||||||
|
return lane
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
t.Fatalf("lane %q not found", id)
|
||||||
|
return pipeline.ResolvedArtifactLane{}
|
||||||
|
}
|
||||||
|
|
||||||
|
func referenceContractBindingSource(bindings []pipeline.ReferenceBinding, slot string) string {
|
||||||
|
for _, binding := range bindings {
|
||||||
|
if binding.SlotName == slot {
|
||||||
|
return binding.Source
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
|
||||||
|
func referenceContractFindBinding(bindings []pipeline.ReferenceBinding, slot string) *pipeline.ReferenceBinding {
|
||||||
|
for i := range bindings {
|
||||||
|
if bindings[i].SlotName == slot {
|
||||||
|
return &bindings[i]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func referenceContractFileURI(path string) string {
|
||||||
|
absolute, err := filepath.Abs(path)
|
||||||
|
if err != nil {
|
||||||
|
absolute = path
|
||||||
|
}
|
||||||
|
return "file://" + filepath.ToSlash(absolute)
|
||||||
|
}
|
||||||
1217
internal/cli/run.go
1217
internal/cli/run.go
File diff suppressed because it is too large
Load Diff
501
internal/cli/run_contract_test.go
Normal file
501
internal/cli/run_contract_test.go
Normal file
@@ -0,0 +1,501 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestRunControlsRejectSyntaxWithoutAllocatingState(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
args func(stateTestRoots) []string
|
||||||
|
}{
|
||||||
|
{name: "missing pipeline", args: func(roots stateTestRoots) []string {
|
||||||
|
return []string{"run", "--config", roots.config, "--input", roots.input}
|
||||||
|
}},
|
||||||
|
{name: "missing input", args: func(roots stateTestRoots) []string {
|
||||||
|
return []string{"run", "sample", "--config", roots.config}
|
||||||
|
}},
|
||||||
|
{name: "unknown flag", args: func(roots stateTestRoots) []string {
|
||||||
|
return []string{"run", "sample", "--config", roots.config, "--input", roots.input, "--unknown"}
|
||||||
|
}},
|
||||||
|
{name: "blank output directory", args: func(roots stateTestRoots) []string {
|
||||||
|
return []string{"run", "sample", "--config", roots.config, "--input", roots.input, "--output-dir", ""}
|
||||||
|
}},
|
||||||
|
{name: "blank debug directory", args: func(roots stateTestRoots) []string {
|
||||||
|
return []string{"run", "sample", "--config", roots.config, "--input", roots.input, "--debug-dir", ""}
|
||||||
|
}},
|
||||||
|
{name: "debug directory without debug", args: func(roots stateTestRoots) []string {
|
||||||
|
return []string{"run", "sample", "--config", roots.config, "--input", roots.input, "--debug-dir", filepath.Join(filepath.Dir(roots.debug), "requested-debug")}
|
||||||
|
}},
|
||||||
|
{name: "blank session ID", args: func(roots stateTestRoots) []string {
|
||||||
|
return []string{"run", "sample", "--config", roots.config, "--input", roots.input, "--session-id", ""}
|
||||||
|
}},
|
||||||
|
{name: "multiple pipeline IDs", args: func(roots stateTestRoots) []string {
|
||||||
|
return []string{"run", "sample", "extra", "--config", roots.config, "--input", roots.input}
|
||||||
|
}},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions(tt.args(roots), &stdout, &stderr, newStateTestHarness().options())
|
||||||
|
if code != 2 || stdout.Len() != 0 || stderr.Len() == 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
assertAbsent(t, roots.output)
|
||||||
|
assertAbsent(t, roots.debug)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRecomputeStepCLIContract(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
configure func(*testing.T, stateTestRoots)
|
||||||
|
flags []string
|
||||||
|
wantCode int
|
||||||
|
wantOutput string
|
||||||
|
wantError string
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
name: "explicit step",
|
||||||
|
configure: func(t *testing.T, roots stateTestRoots) {
|
||||||
|
replaceStateTestConfigLine(t, roots.config, " artifacts:\n items:\n extract: test/extract\n merge: test/merge\n normalize: test/normalize\n", " steps:\n - id: chosen\n artifacts:\n items:\n extract: test/extract\n merge: test/merge\n normalize: test/normalize\n")
|
||||||
|
},
|
||||||
|
flags: []string{"--resume", "--recompute-step", "chosen"},
|
||||||
|
wantCode: 0,
|
||||||
|
wantOutput: "outputs=1",
|
||||||
|
},
|
||||||
|
{name: "implicit default step", flags: []string{"--resume", "--recompute-step", "default"}, wantCode: 0, wantOutput: "outputs=1"},
|
||||||
|
{name: "repeated flag", flags: []string{"--resume", "--recompute-step", "default", "--recompute-step", "default"}, wantCode: 2, wantError: "specified only once"},
|
||||||
|
{name: "empty step", flags: []string{"--resume", "--recompute-step", ""}, wantCode: 2, wantError: "must not be empty"},
|
||||||
|
{name: "unknown step", flags: []string{"--resume", "--recompute-step", "missing"}, wantCode: 1, wantError: "unknown pipeline step"},
|
||||||
|
{name: "without resume", flags: []string{"--recompute-step", "default"}, wantCode: 2, wantError: "requires --resume"},
|
||||||
|
{
|
||||||
|
name: "checkpoint recording disabled",
|
||||||
|
configure: func(t *testing.T, roots stateTestRoots) {
|
||||||
|
replaceStateTestConfigLine(t, roots.config, " enabled: true\n", " enabled: false\n")
|
||||||
|
},
|
||||||
|
flags: []string{"--resume", "--recompute-step", "default"}, wantCode: 1, wantError: "cache.checkpoints.enabled",
|
||||||
|
},
|
||||||
|
{name: "with only", flags: []string{"--resume", "--recompute-step", "default", "--only", "items"}, wantCode: 2, wantError: "cannot be combined with --only"},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
if tt.configure != nil {
|
||||||
|
tt.configure(t, roots)
|
||||||
|
}
|
||||||
|
args := []string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass"}
|
||||||
|
args = append(args, tt.flags...)
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions(args, &stdout, &stderr, newStateTestHarness().options())
|
||||||
|
if code != tt.wantCode || (tt.wantOutput != "" && !strings.Contains(stdout.String(), tt.wantOutput)) || (tt.wantError != "" && !strings.Contains(stderr.String(), tt.wantError)) {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunValidFailuresClassifyAndReportDebug(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
args func(stateTestRoots) []string
|
||||||
|
wantError string
|
||||||
|
wantDebug bool
|
||||||
|
}{
|
||||||
|
{name: "unknown pipeline", args: func(roots stateTestRoots) []string {
|
||||||
|
return []string{"run", "missing", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass"}
|
||||||
|
}, wantError: `pipeline "missing"`},
|
||||||
|
{name: "unknown lane", args: func(roots stateTestRoots) []string {
|
||||||
|
return []string{"run", "sample", "--config", roots.config, "--input", roots.input, "--only", "missing", "--chunk_cache", "bypass", "--debug"}
|
||||||
|
}, wantError: `lane "missing"`, wantDebug: true},
|
||||||
|
{name: "unreadable input", args: func(roots stateTestRoots) []string {
|
||||||
|
return []string{"run", "sample", "--config", roots.config, "--input", filepath.Join(filepath.Dir(roots.input), "unreadable.txt"), "--chunk_cache", "bypass", "--debug"}
|
||||||
|
}, wantError: "read input", wantDebug: true},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions(tt.args(roots), &stdout, &stderr, newStateTestHarness().options())
|
||||||
|
if code != 1 || stdout.Len() != 0 || !strings.Contains(stderr.String(), tt.wantError) {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
if tt.wantDebug {
|
||||||
|
if !strings.Contains(stderr.String(), "debug=") {
|
||||||
|
t.Fatalf("stderr=%q, want debug path", stderr.String())
|
||||||
|
}
|
||||||
|
onlyChildDir(t, roots.debug)
|
||||||
|
} else {
|
||||||
|
assertAbsent(t, roots.debug)
|
||||||
|
}
|
||||||
|
assertAbsent(t, roots.output)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunOnlyExecutesSelectedLanes(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
data, err := os.ReadFile(roots.config)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
data = []byte(replaceRequiredOnce(t, string(data), " output: test/output\n", " other:\n extract: test/extract\n output: test/output\n"))
|
||||||
|
if err := os.WriteFile(roots.config, data, 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--only", "items", "--chunk_cache", "bypass"}, &stdout, &stderr, harness.options())
|
||||||
|
if code != 0 || !strings.Contains(stdout.String(), "outputs=1") || stderr.Len() != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
harness.mu.Lock()
|
||||||
|
extractCalls := harness.extractCalls
|
||||||
|
harness.mu.Unlock()
|
||||||
|
if extractCalls != 1 {
|
||||||
|
t.Fatalf("extract calls = %d, want only the selected lane", extractCalls)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunStateRootsHonorEnvironmentFlagsAndDefaults(t *testing.T) {
|
||||||
|
t.Run("environment roots", func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
environmentOutput := filepath.Join(t.TempDir(), "environment-output")
|
||||||
|
environmentDebug := filepath.Join(t.TempDir(), "environment-debug")
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
opts.LookupEnv = lookupRunContractEnv(map[string]string{
|
||||||
|
"NOTARIUS_OUTPUT_DIR": environmentOutput,
|
||||||
|
"NOTARIUS_DEBUG_DIR": environmentDebug,
|
||||||
|
})
|
||||||
|
result := runWithStateRoots(t, roots, opts, nil)
|
||||||
|
if result.code != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", result.code, result.stdout, result.stderr)
|
||||||
|
}
|
||||||
|
assertFile(t, filepath.Join(environmentOutput, filepath.Base(onlyChildDir(t, environmentOutput)), "result.json"))
|
||||||
|
onlyChildDir(t, environmentDebug)
|
||||||
|
assertAbsent(t, roots.output)
|
||||||
|
assertAbsent(t, roots.debug)
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("command flags override environment", func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
environmentOutput := filepath.Join(t.TempDir(), "environment-output")
|
||||||
|
environmentDebug := filepath.Join(t.TempDir(), "environment-debug")
|
||||||
|
flagOutput := filepath.Join(t.TempDir(), "flag-output")
|
||||||
|
flagDebug := filepath.Join(t.TempDir(), "flag-debug")
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
opts.LookupEnv = lookupRunContractEnv(map[string]string{
|
||||||
|
"NOTARIUS_OUTPUT_DIR": environmentOutput,
|
||||||
|
"NOTARIUS_DEBUG_DIR": environmentDebug,
|
||||||
|
})
|
||||||
|
result := runWithStateRoots(t, roots, opts, []string{"--output-dir", flagOutput, "--debug-dir", flagDebug})
|
||||||
|
if result.code != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", result.code, result.stdout, result.stderr)
|
||||||
|
}
|
||||||
|
assertFile(t, filepath.Join(flagOutput, filepath.Base(onlyChildDir(t, flagOutput)), "result.json"))
|
||||||
|
onlyChildDir(t, flagDebug)
|
||||||
|
assertAbsent(t, environmentOutput)
|
||||||
|
assertAbsent(t, environmentDebug)
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("built-in roots", func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
data, err := os.ReadFile(roots.config)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
text := string(data)
|
||||||
|
text = replaceRequiredOnce(t, text, fmt.Sprintf(" directory: %q\n", roots.output), "")
|
||||||
|
text = replaceRequiredOnce(t, text, fmt.Sprintf(" directory: %q\n", roots.debug), "")
|
||||||
|
if err := os.WriteFile(roots.config, []byte(text), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
workDir := t.TempDir()
|
||||||
|
t.Chdir(workDir)
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
result := runWithStateRoots(t, roots, opts, nil)
|
||||||
|
if result.code != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", result.code, result.stdout, result.stderr)
|
||||||
|
}
|
||||||
|
assertFile(t, filepath.Join(workDir, "notarius-output", filepath.Base(onlyChildDir(t, filepath.Join(workDir, "notarius-output"))), "result.json"))
|
||||||
|
onlyChildDir(t, filepath.Join(workDir, "notarius-debug"))
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunLLMProfileOverrideAndValidationUseInjectedBoundaries(t *testing.T) {
|
||||||
|
t.Run("one effective profile reaches the factory and modules", func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
profileDir := writeRunContractProfiles(t, "override-profile")
|
||||||
|
prependRunContractConfig(t, roots, fmt.Sprintf("scriptorium:\n profile_dir: %q\n", profileDir))
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
var factoryProfiles []string
|
||||||
|
opts := harness.options()
|
||||||
|
opts.LLMClientFactory = func(_ context.Context, _ config.Config, profileID string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||||
|
factoryProfiles = append(factoryProfiles, profileID)
|
||||||
|
return nil, nil, nil
|
||||||
|
}
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass", "--llm-profile", "override-profile"}, &stdout, &stderr, opts)
|
||||||
|
if code != 0 || stderr.Len() != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
if len(factoryProfiles) != 1 || factoryProfiles[0] != "override-profile" {
|
||||||
|
t.Fatalf("factory profiles = %#v, want one override profile", factoryProfiles)
|
||||||
|
}
|
||||||
|
harness.mu.Lock()
|
||||||
|
profiles := append([]string(nil), harness.moduleProfiles...)
|
||||||
|
harness.mu.Unlock()
|
||||||
|
if len(profiles) < 4 {
|
||||||
|
t.Fatalf("module profiles = %#v, want chunk and lane stage requests", profiles)
|
||||||
|
}
|
||||||
|
for _, profile := range profiles {
|
||||||
|
if profile != "override-profile" {
|
||||||
|
t.Fatalf("module profiles = %#v, want override on every request", profiles)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("validator profile remains distinct", func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
profileDir := writeRunContractProfiles(t, "override-profile", "validator-profile")
|
||||||
|
prependRunContractConfig(t, roots, fmt.Sprintf("scriptorium:\n profile_dir: %q\n", profileDir))
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
var validatorProfiles []string
|
||||||
|
opts := harness.options()
|
||||||
|
registerRunContractValidator(t, &opts, &validatorProfiles)
|
||||||
|
factoryProfiles := []string{}
|
||||||
|
opts.LLMClientFactory = func(_ context.Context, _ config.Config, profileID string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||||
|
factoryProfiles = append(factoryProfiles, profileID)
|
||||||
|
return nil, nil, nil
|
||||||
|
}
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass", "--llm-profile", "override-profile"}, &stdout, &stderr, opts)
|
||||||
|
if code != 0 || stderr.Len() != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
if len(factoryProfiles) != 1 || factoryProfiles[0] != "" {
|
||||||
|
t.Fatalf("factory profiles = %#v, want one call without a unique profile", factoryProfiles)
|
||||||
|
}
|
||||||
|
if len(validatorProfiles) != 1 || validatorProfiles[0] != "validator-profile" {
|
||||||
|
t.Fatalf("validator profiles = %#v, want configured validator profile", validatorProfiles)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("unknown profile is rejected without factory access", func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
profileDir := writeRunContractProfiles(t, "override-profile")
|
||||||
|
prependRunContractConfig(t, roots, fmt.Sprintf("scriptorium:\n profile_dir: %q\n", profileDir))
|
||||||
|
factoryCalls := 0
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
opts.LLMClientFactory = func(context.Context, config.Config, string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||||
|
factoryCalls++
|
||||||
|
return nil, nil, nil
|
||||||
|
}
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass", "--llm-profile", "missing-profile"}, &stdout, &stderr, opts)
|
||||||
|
if code != 1 || !strings.Contains(stderr.String(), "not configured") || factoryCalls != 0 || stdout.Len() != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q factoryCalls=%d", code, stdout.String(), stderr.String(), factoryCalls)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEffectiveLLMProfileIDsAreSortedDeduplicatedAndLLMOnly(t *testing.T) {
|
||||||
|
resolved := pipeline.ResolvedPipeline{
|
||||||
|
Input: pipeline.ModuleBinding{LLMProfile: "input-profile"},
|
||||||
|
Chunk: pipeline.ModuleBinding{LLMProfile: " zeta "},
|
||||||
|
Steps: []pipeline.ResolvedPipelineStep{{
|
||||||
|
ID: "default",
|
||||||
|
ArtifactLanes: []pipeline.ResolvedArtifactLane{{
|
||||||
|
Extract: pipeline.ModuleBinding{LLMProfile: "alpha"},
|
||||||
|
Merge: pipeline.ModuleBinding{LLMProfile: "zeta"},
|
||||||
|
Normalize: pipeline.ModuleBinding{LLMProfile: " gamma "},
|
||||||
|
}},
|
||||||
|
}},
|
||||||
|
ValidatorChains: []pipeline.ResolvedValidatorChain{{Validators: []pipeline.ResolvedValidator{
|
||||||
|
{Binding: pipeline.ModuleBinding{LLMProfile: "deterministic-profile"}, ExecutionClass: contracts.ExecutionClassDeterministic},
|
||||||
|
{Binding: pipeline.ModuleBinding{LLMProfile: "beta"}, ExecutionClass: contracts.ExecutionClassLLMBacked},
|
||||||
|
}}},
|
||||||
|
Output: pipeline.ModuleBinding{LLMProfile: "output-profile"},
|
||||||
|
}
|
||||||
|
got := effectiveLLMProfileIDs(resolved)
|
||||||
|
want := []string{"alpha", "beta", "gamma", "zeta"}
|
||||||
|
if strings.Join(got, ",") != strings.Join(want, ",") {
|
||||||
|
t.Fatalf("effective profiles = %#v, want %#v", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunSessionIDUsesExplicitValueOrSourceDocumentID(t *testing.T) {
|
||||||
|
for _, tt := range []struct {
|
||||||
|
name string
|
||||||
|
args []string
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{name: "source document", want: "source"},
|
||||||
|
{name: "explicit trimmed value", args: []string{"--session-id", " explicit-session "}, want: "explicit-session"},
|
||||||
|
} {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
args := append([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass"}, tt.args...)
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions(args, &stdout, &stderr, harness.options())
|
||||||
|
if code != 0 || stderr.Len() != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
harness.mu.Lock()
|
||||||
|
sessions := append([]string(nil), harness.sessionIDs...)
|
||||||
|
harness.mu.Unlock()
|
||||||
|
if len(sessions) < 4 {
|
||||||
|
t.Fatalf("session IDs = %#v, want all prompt-facing module requests", sessions)
|
||||||
|
}
|
||||||
|
for _, session := range sessions {
|
||||||
|
if session != tt.want {
|
||||||
|
t.Fatalf("session IDs = %#v, want %q", sessions, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunFactoryAndPreparationFailuresAreProcessFailures(t *testing.T) {
|
||||||
|
t.Run("LLM factory", func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
opts.LLMClientFactory = func(context.Context, config.Config, string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||||
|
return nil, nil, errors.New("injected LLM factory failure")
|
||||||
|
}
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass"}, &stdout, &stderr, opts)
|
||||||
|
if code != 1 || !strings.Contains(stderr.String(), "injected LLM factory failure") || stdout.Len() != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("pipeline preparation", func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
data, err := os.ReadFile(roots.config)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
data = []byte(replaceRequiredOnce(t, string(data), "extract: test/extract", "extract: test/failing-extract"))
|
||||||
|
if err := os.WriteFile(roots.config, data, 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
if err := pipeline.RegisterExtractorBuilder(opts.Registries.Extractors, pipeline.ModuleSpec{Key: "test/failing-extract", Stage: pipeline.StageExtract, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: stateTestArtifactKind}, func(map[string]any) error { return nil }, func(pipeline.BuildRequest) (contracts.Extractor[stateTestArtifact], error) {
|
||||||
|
return nil, errors.New("injected extractor construction failure")
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
opts.Catalog = catalogFromRegistries(opts.Registries)
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass"}, &stdout, &stderr, opts)
|
||||||
|
if code != 1 || !strings.Contains(stderr.String(), "injected extractor construction failure") || stdout.Len() != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunWarningsRemainSuccessfulAndReachDurableSurfaces(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
harness.includeWarnings = true
|
||||||
|
harness.chunkWarnings = []contracts.Warning{{Scope: "chunk", ReasonCode: "contract-warning", Message: "warning retained"}}
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass", "--debug"}, &stdout, &stderr, harness.options())
|
||||||
|
if code != 0 || !strings.Contains(stdout.String(), "outputs=1") || !strings.Contains(stderr.String(), "1 warning(s)") {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
outputPath := filepath.Join(onlyChildDir(t, roots.output), "result.json")
|
||||||
|
output, err := os.ReadFile(outputPath)
|
||||||
|
if err != nil || !strings.Contains(string(output), "contract-warning") {
|
||||||
|
t.Fatalf("durable output = %q, %v", output, err)
|
||||||
|
}
|
||||||
|
bundle := onlyChildDir(t, roots.debug)
|
||||||
|
var warnings []contracts.Warning
|
||||||
|
readStateTestSummaryJSON(t, bundle, "warnings.json", &warnings)
|
||||||
|
if len(warnings) != 1 || warnings[0].ReasonCode != "contract-warning" {
|
||||||
|
t.Fatalf("debug warnings = %#v", warnings)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func runWithStateRoots(t *testing.T, roots stateTestRoots, opts Options, extra []string) stateTestResult {
|
||||||
|
t.Helper()
|
||||||
|
args := []string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass", "--debug"}
|
||||||
|
args = append(args, extra...)
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
return stateTestResult{code: RunWithOptions(args, &stdout, &stderr, opts), stdout: stdout.String(), stderr: stderr.String()}
|
||||||
|
}
|
||||||
|
|
||||||
|
func lookupRunContractEnv(values map[string]string) func(string) (string, bool) {
|
||||||
|
return func(name string) (string, bool) {
|
||||||
|
value, ok := values[name]
|
||||||
|
return value, ok
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func prependRunContractConfig(t *testing.T, roots stateTestRoots, prefix string) {
|
||||||
|
t.Helper()
|
||||||
|
data, err := os.ReadFile(roots.config)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(roots.config, append([]byte(prefix), data...), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func writeRunContractProfiles(t *testing.T, ids ...string) string {
|
||||||
|
t.Helper()
|
||||||
|
dir := t.TempDir()
|
||||||
|
for _, id := range ids {
|
||||||
|
profile := fmt.Sprintf("id: %s\nendpoint: http://127.0.0.1:1/v1\nmodel: %s-model\n", id, id)
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, id+".yaml"), []byte(profile), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return dir
|
||||||
|
}
|
||||||
|
|
||||||
|
func registerRunContractValidator(t *testing.T, opts *Options, profiles *[]string) {
|
||||||
|
t.Helper()
|
||||||
|
if err := pipeline.RegisterTypedValidatorBuilder(opts.Registries.Validators, stateTestArtifactKind, pipeline.ValidatorSpec{Key: "run-contract-validator", ExecutionClass: contracts.ExecutionClassLLMBacked}, func(map[string]any) error { return nil }, func(pipeline.BuildRequest) (contracts.TypedValidator[stateTestArtifact], error) {
|
||||||
|
return runContractValidator{profiles: profiles}, nil
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := opts.Registries.ValidatorChains.Register(pipeline.ValidatorChainMapping{Stage: pipeline.StageExtract, Module: "test/extract", Validators: []pipeline.ModuleBinding{{Module: "run-contract-validator", LLMProfile: "validator-profile"}}}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
opts.Catalog = catalogFromRegistries(opts.Registries)
|
||||||
|
}
|
||||||
|
|
||||||
|
type runContractValidator struct {
|
||||||
|
profiles *[]string
|
||||||
|
}
|
||||||
|
|
||||||
|
func (v runContractValidator) Name() string { return "run-contract-validator" }
|
||||||
|
|
||||||
|
func (v runContractValidator) ExecutionClass() contracts.ExecutionClass {
|
||||||
|
return contracts.ExecutionClassLLMBacked
|
||||||
|
}
|
||||||
|
|
||||||
|
func (v runContractValidator) Validate(_ context.Context, req contracts.TypedValidationRequest[stateTestArtifact]) (contracts.ValidationResult, error) {
|
||||||
|
*v.profiles = append(*v.profiles, req.LLMProfile)
|
||||||
|
return contracts.ValidationResult{Approved: true}, nil
|
||||||
|
}
|
||||||
34
internal/cli/run_id.go
Normal file
34
internal/cli/run_id.go
Normal file
@@ -0,0 +1,34 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"crypto/rand"
|
||||||
|
"encoding/hex"
|
||||||
|
"fmt"
|
||||||
|
"io"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
type RunIDGenerator func(time.Time) (string, error)
|
||||||
|
|
||||||
|
func defaultRunIDGenerator(startedAt time.Time) (string, error) {
|
||||||
|
var suffix [16]byte
|
||||||
|
if _, err := io.ReadFull(rand.Reader, suffix[:]); err != nil {
|
||||||
|
return "", fmt.Errorf("read random run ID suffix: %w", err)
|
||||||
|
}
|
||||||
|
return fmt.Sprintf("run-%d-%s", startedAt.UnixNano(), hex.EncodeToString(suffix[:])), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func validateRunID(runID string) error {
|
||||||
|
if runID == "" {
|
||||||
|
return fmt.Errorf("run ID must not be empty")
|
||||||
|
}
|
||||||
|
if runID != strings.TrimSpace(runID) {
|
||||||
|
return fmt.Errorf("run ID %q must not have surrounding whitespace", runID)
|
||||||
|
}
|
||||||
|
if strings.ContainsAny(runID, `/\\`) || filepath.IsAbs(runID) || filepath.Clean(runID) != runID || runID == "." || runID == ".." {
|
||||||
|
return fmt.Errorf("run ID %q must be one safe path component", runID)
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
87
internal/cli/run_id_test.go
Normal file
87
internal/cli/run_id_test.go
Normal file
@@ -0,0 +1,87 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"regexp"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestDefaultRunIDGeneratorProducesUniqueSafeIDs(t *testing.T) {
|
||||||
|
startedAt := time.Unix(0, 123456789).UTC()
|
||||||
|
pattern := regexp.MustCompile(`^run-123456789-[0-9a-f]{32}$`)
|
||||||
|
seen := make(map[string]struct{}, 256)
|
||||||
|
for i := 0; i < 256; i++ {
|
||||||
|
runID, err := defaultRunIDGenerator(startedAt)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if !pattern.MatchString(runID) {
|
||||||
|
t.Fatalf("run ID %q does not match production format", runID)
|
||||||
|
}
|
||||||
|
if err := validateRunID(runID); err != nil {
|
||||||
|
t.Fatalf("run ID %q is not path-safe: %v", runID, err)
|
||||||
|
}
|
||||||
|
if _, exists := seen[runID]; exists {
|
||||||
|
t.Fatalf("duplicate run ID %q", runID)
|
||||||
|
}
|
||||||
|
seen[runID] = struct{}{}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestWriteOutputFilesSupportsNestedLogicalPaths(t *testing.T) {
|
||||||
|
runPath := filepath.Join(t.TempDir(), "output", "run-safe")
|
||||||
|
if err := writeOutputFiles(runPath, []contracts.OutputFile{{Name: "nested/result.json", Bytes: []byte("result")}}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
data, err := os.ReadFile(filepath.Join(runPath, "nested", "result.json"))
|
||||||
|
if err != nil || string(data) != "result" {
|
||||||
|
t.Fatalf("nested output = %q, %v", data, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestWriteOutputFilesRejectsUnsafeNamesBeforeAllocatingRunDirectory(t *testing.T) {
|
||||||
|
outputRoot := filepath.Join(t.TempDir(), "output")
|
||||||
|
runPath := filepath.Join(outputRoot, "run-safe")
|
||||||
|
for _, name := range []string{"", "../outside", "/absolute", `nested\\outside`, "nested/../outside"} {
|
||||||
|
t.Run(name, func(t *testing.T) {
|
||||||
|
if err := writeOutputFiles(runPath, []contracts.OutputFile{{Name: "safe.json"}, {Name: name}}); err == nil {
|
||||||
|
t.Fatalf("writeOutputFiles accepted %q", name)
|
||||||
|
}
|
||||||
|
if _, err := os.Stat(outputRoot); !os.IsNotExist(err) {
|
||||||
|
t.Fatalf("output root exists or stat failed after %q: %v", name, err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestWriteOutputFilesRetainsNewPartialDirectoryAndPreservesSibling(t *testing.T) {
|
||||||
|
outputRoot := filepath.Join(t.TempDir(), "output")
|
||||||
|
siblingPath := filepath.Join(outputRoot, "sibling")
|
||||||
|
if err := os.MkdirAll(siblingPath, 0o755); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
sentinelPath := filepath.Join(siblingPath, "sentinel")
|
||||||
|
if err := os.WriteFile(sentinelPath, []byte("preserve sibling"), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
runPath := filepath.Join(outputRoot, "run-safe")
|
||||||
|
err := writeOutputFiles(runPath, []contracts.OutputFile{
|
||||||
|
{Name: "blocked", Bytes: []byte("partial output")},
|
||||||
|
{Name: "blocked/nested.json", Bytes: []byte("unreachable")},
|
||||||
|
})
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "create output directory") {
|
||||||
|
t.Fatalf("writeOutputFiles() error = %v, want later directory failure", err)
|
||||||
|
}
|
||||||
|
if got, err := os.ReadFile(filepath.Join(runPath, "blocked")); err != nil || string(got) != "partial output" {
|
||||||
|
t.Fatalf("partial output = %q, %v", got, err)
|
||||||
|
}
|
||||||
|
if got, err := os.ReadFile(sentinelPath); err != nil || string(got) != "preserve sibling" {
|
||||||
|
t.Fatalf("sibling sentinel = %q, %v", got, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
111
internal/cli/run_result.go
Normal file
111
internal/cli/run_result.go
Normal file
@@ -0,0 +1,111 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
|
"io"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
)
|
||||||
|
|
||||||
|
const runResultSchemaVersion = "notarius.run-result.v1"
|
||||||
|
|
||||||
|
type runResult struct {
|
||||||
|
SchemaVersion string `json:"schema_version"`
|
||||||
|
RunID string `json:"run_id"`
|
||||||
|
PipelineID string `json:"pipeline_id"`
|
||||||
|
OutputDirectory string `json:"output_directory"`
|
||||||
|
IndexFile string `json:"index_file,omitempty"`
|
||||||
|
NormalizedOutputCount int `json:"normalized_output_count"`
|
||||||
|
RejectedOutputCount int `json:"rejected_output_count"`
|
||||||
|
WarningCount int `json:"warning_count"`
|
||||||
|
ValidationStatus string `json:"validation_status"`
|
||||||
|
DebugDirectory string `json:"debug_directory,omitempty"`
|
||||||
|
}
|
||||||
|
|
||||||
|
func newRunResult(resolved pipeline.ResolvedPipeline, output pipeline.RunOutput, outputDirectory, debugDirectory string) (runResult, error) {
|
||||||
|
if strings.TrimSpace(output.Manifest.RunID) == "" {
|
||||||
|
return runResult{}, fmt.Errorf("run result requires a run ID")
|
||||||
|
}
|
||||||
|
if strings.TrimSpace(resolved.ID) == "" {
|
||||||
|
return runResult{}, fmt.Errorf("run result requires a resolved pipeline ID")
|
||||||
|
}
|
||||||
|
if strings.TrimSpace(output.Manifest.PipelineID) == "" {
|
||||||
|
return runResult{}, fmt.Errorf("run result requires a manifest pipeline ID")
|
||||||
|
}
|
||||||
|
if output.Manifest.PipelineID != resolved.ID {
|
||||||
|
return runResult{}, fmt.Errorf("run result pipeline ID does not match resolved pipeline")
|
||||||
|
}
|
||||||
|
if strings.TrimSpace(output.Manifest.ValidationStatus) == "" {
|
||||||
|
return runResult{}, fmt.Errorf("run result requires a validation status")
|
||||||
|
}
|
||||||
|
if strings.TrimSpace(outputDirectory) == "" {
|
||||||
|
return runResult{}, fmt.Errorf("run result requires an output directory")
|
||||||
|
}
|
||||||
|
|
||||||
|
absOutputDirectory, err := filepath.Abs(outputDirectory)
|
||||||
|
if err != nil {
|
||||||
|
return runResult{}, fmt.Errorf("make output directory absolute: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
result := runResult{
|
||||||
|
SchemaVersion: runResultSchemaVersion,
|
||||||
|
RunID: output.Manifest.RunID,
|
||||||
|
PipelineID: output.Manifest.PipelineID,
|
||||||
|
OutputDirectory: absOutputDirectory,
|
||||||
|
NormalizedOutputCount: len(output.NormalizeOutputs),
|
||||||
|
RejectedOutputCount: len(output.Rejected),
|
||||||
|
WarningCount: len(output.Warnings),
|
||||||
|
ValidationStatus: output.Manifest.ValidationStatus,
|
||||||
|
}
|
||||||
|
|
||||||
|
if strings.TrimSpace(debugDirectory) != "" {
|
||||||
|
absDebugDirectory, err := filepath.Abs(debugDirectory)
|
||||||
|
if err != nil {
|
||||||
|
return runResult{}, fmt.Errorf("make debug directory absolute: %w", err)
|
||||||
|
}
|
||||||
|
result.DebugDirectory = absDebugDirectory
|
||||||
|
}
|
||||||
|
|
||||||
|
if resolved.Output.Module == pipeline.DefaultOutputModule {
|
||||||
|
indexCount := 0
|
||||||
|
for _, file := range output.OutputFiles {
|
||||||
|
if file.Name == "index.json" {
|
||||||
|
indexCount++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if indexCount != 1 {
|
||||||
|
return runResult{}, fmt.Errorf("production JSON output must contain exactly one index.json file")
|
||||||
|
}
|
||||||
|
result.IndexFile = "index.json"
|
||||||
|
}
|
||||||
|
|
||||||
|
return result, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func encodeRunResult(result runResult) ([]byte, error) {
|
||||||
|
encoded, err := json.Marshal(result)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("encode run result: %w", err)
|
||||||
|
}
|
||||||
|
return append(encoded, '\n'), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func writeRunResult(writer io.Writer, content []byte) error {
|
||||||
|
for len(content) > 0 {
|
||||||
|
written, err := writer.Write(content)
|
||||||
|
if written < 0 || written > len(content) {
|
||||||
|
return io.ErrShortWrite
|
||||||
|
}
|
||||||
|
content = content[written:]
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if written == 0 {
|
||||||
|
return io.ErrShortWrite
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
199
internal/cli/run_result_command_test.go
Normal file
199
internal/cli/run_result_command_test.go
Normal file
@@ -0,0 +1,199 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"encoding/json"
|
||||||
|
"errors"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||||
|
alwaysreject "gitea.maximumdirect.net/eric/notarius/internal/modules/generic/validate/always_reject"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestMaintainedMinimalInvocationEmitsRunResult(t *testing.T) {
|
||||||
|
outputRoot := filepath.Join(t.TempDir(), "output")
|
||||||
|
var stdout, stderr strings.Builder
|
||||||
|
code := RunWithOptions([]string{
|
||||||
|
"run", "dnd-session",
|
||||||
|
"--config", repositoryPath("examples", "dnd-minimal.config.yml"),
|
||||||
|
"--input", repositoryPath("examples", "seriatim-minimal-transcript.json"),
|
||||||
|
"--only", "spells", "--chunk_cache", "bypass", "--output-dir", outputRoot, "--json",
|
||||||
|
}, &stdout, &stderr, productionRunOptions(t, &productionFakeLLMClient{}))
|
||||||
|
if code != 0 || stderr.Len() != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
receipt := decodeRunResultDocument(t, stdout.String())
|
||||||
|
if got := receipt["schema_version"]; got != "notarius.run-result.v1" {
|
||||||
|
t.Fatalf("schema_version = %q", got)
|
||||||
|
}
|
||||||
|
if got := receipt["run_id"]; got != productionRunID {
|
||||||
|
t.Fatalf("run_id = %q", got)
|
||||||
|
}
|
||||||
|
if got := receipt["pipeline_id"]; got != "dnd-session" {
|
||||||
|
t.Fatalf("pipeline_id = %q", got)
|
||||||
|
}
|
||||||
|
if got := receipt["index_file"]; got != "index.json" {
|
||||||
|
t.Fatalf("index_file = %q", got)
|
||||||
|
}
|
||||||
|
if got := receipt["normalized_output_count"]; got != float64(1) {
|
||||||
|
t.Fatalf("normalized_output_count = %v", got)
|
||||||
|
}
|
||||||
|
if got := receipt["rejected_output_count"]; got != float64(0) {
|
||||||
|
t.Fatalf("rejected_output_count = %v", got)
|
||||||
|
}
|
||||||
|
if got := receipt["warning_count"]; got != float64(0) {
|
||||||
|
t.Fatalf("warning_count = %v", got)
|
||||||
|
}
|
||||||
|
if got := receipt["validation_status"]; got != "approved" {
|
||||||
|
t.Fatalf("validation_status = %q", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
outputDirectory, ok := receipt["output_directory"].(string)
|
||||||
|
if !ok || !filepath.IsAbs(outputDirectory) || outputDirectory != filepath.Join(outputRoot, productionRunID) {
|
||||||
|
t.Fatalf("output_directory = %q", receipt["output_directory"])
|
||||||
|
}
|
||||||
|
indexFile := receipt["index_file"].(string)
|
||||||
|
assertFile(t, filepath.Join(outputDirectory, indexFile))
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunResultReportsWarningsAndDebugBundle(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
harness.chunkWarnings = []contracts.Warning{{Scope: "chunk", ReasonCode: "contract-warning", Message: "warning retained"}}
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{
|
||||||
|
"run", "sample", "--config", roots.config, "--input", roots.input,
|
||||||
|
"--chunk_cache", "bypass", "--debug", "--json",
|
||||||
|
}, &stdout, &stderr, harness.options())
|
||||||
|
if code != 0 || !strings.Contains(stderr.String(), "1 warning(s)") {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
receipt := decodeRunResultDocument(t, stdout.String())
|
||||||
|
if got := receipt["warning_count"]; got != float64(1) {
|
||||||
|
t.Fatalf("warning_count = %v", got)
|
||||||
|
}
|
||||||
|
debugDirectory, ok := receipt["debug_directory"].(string)
|
||||||
|
if !ok || !filepath.IsAbs(debugDirectory) || debugDirectory != onlyChildDir(t, roots.debug) {
|
||||||
|
t.Fatalf("debug_directory = %q", receipt["debug_directory"])
|
||||||
|
}
|
||||||
|
if strings.Contains(stdout.String(), "complete:") || strings.Contains(stdout.String(), "debug=") {
|
||||||
|
t.Fatalf("machine stdout contains human reporting: %q", stdout.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunResultReportsSuccessfulRejection(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
configBytes, err := os.ReadFile(roots.config)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
configBytes = []byte(replaceRequiredOnce(t, string(configBytes), " normalize: test/normalize\n", " normalize:\n module: test/normalize\n validators:\n - generic/always_reject\n"))
|
||||||
|
if err := os.WriteFile(roots.config, configBytes, 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
opts := harness.options()
|
||||||
|
if err := alwaysreject.RegisterTyped[stateTestArtifact](opts.Registries.Validators, stateTestArtifactKind); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
opts.Catalog = catalogFromRegistries(opts.Registries)
|
||||||
|
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{
|
||||||
|
"run", "sample", "--config", roots.config, "--input", roots.input,
|
||||||
|
"--chunk_cache", "bypass", "--json",
|
||||||
|
}, &stdout, &stderr, opts)
|
||||||
|
if code != 0 || stderr.Len() != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
receipt := decodeRunResultDocument(t, stdout.String())
|
||||||
|
if got := receipt["normalized_output_count"]; got != float64(0) {
|
||||||
|
t.Fatalf("normalized_output_count = %v", got)
|
||||||
|
}
|
||||||
|
if got := receipt["rejected_output_count"]; got != float64(1) {
|
||||||
|
t.Fatalf("rejected_output_count = %v", got)
|
||||||
|
}
|
||||||
|
if got := receipt["validation_status"]; got != "rejected" {
|
||||||
|
t.Fatalf("validation_status = %q", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunResultIsAbsentForSyntaxAndRuntimeFailures(t *testing.T) {
|
||||||
|
t.Run("syntax", func(t *testing.T) {
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"run", "sample", "--json"}, &stdout, &stderr, newStateTestHarness().options())
|
||||||
|
if code != 2 || stdout.Len() != 0 || stderr.Len() == 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("runtime", func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
harness.extractErr = errors.New("injected extraction failure")
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{
|
||||||
|
"run", "sample", "--config", roots.config, "--input", roots.input,
|
||||||
|
"--chunk_cache", "bypass", "--json",
|
||||||
|
}, &stdout, &stderr, harness.options())
|
||||||
|
if code != 1 || stdout.Len() != 0 || stderr.Len() == 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunResultDeliveryFailureRetainsPublishedBundles(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
writerErr := errors.New("result writer sentinel")
|
||||||
|
stdout := &resultDeliveryWriter{err: writerErr}
|
||||||
|
var stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{
|
||||||
|
"run", "sample", "--config", roots.config, "--input", roots.input,
|
||||||
|
"--chunk_cache", "bypass", "--debug", "--json",
|
||||||
|
}, stdout, &stderr, newStateTestHarness().options())
|
||||||
|
if code != 1 || !strings.Contains(stderr.String(), "write run result") || strings.Contains(stderr.String(), writerErr.Error()) {
|
||||||
|
t.Fatalf("code=%d stderr=%q", code, stderr.String())
|
||||||
|
}
|
||||||
|
if stdout.accepted.Len() != 0 {
|
||||||
|
t.Fatalf("accepted stdout = %q", stdout.accepted.String())
|
||||||
|
}
|
||||||
|
assertStateTestOutput(t, roots.output)
|
||||||
|
debugBundle := onlyChildDir(t, roots.debug)
|
||||||
|
report := readStateTestRunReport(t, debugBundle)
|
||||||
|
if !report.Succeeded {
|
||||||
|
t.Fatalf("debug report = %#v, want successful persisted run", report)
|
||||||
|
}
|
||||||
|
if strings.Contains(readAllFiles(t, debugBundle), writerErr.Error()) {
|
||||||
|
t.Fatalf("debug bundle contains result writer error")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func decodeRunResultDocument(t *testing.T, stdout string) map[string]any {
|
||||||
|
t.Helper()
|
||||||
|
if strings.Count(stdout, "\n") != 1 {
|
||||||
|
t.Fatalf("stdout = %q, want one JSON document", stdout)
|
||||||
|
}
|
||||||
|
var receipt map[string]any
|
||||||
|
if err := json.Unmarshal([]byte(stdout), &receipt); err != nil {
|
||||||
|
t.Fatalf("decode run result: %v; stdout=%q", err, stdout)
|
||||||
|
}
|
||||||
|
return receipt
|
||||||
|
}
|
||||||
|
|
||||||
|
type resultDeliveryWriter struct {
|
||||||
|
err error
|
||||||
|
accepted bytes.Buffer
|
||||||
|
}
|
||||||
|
|
||||||
|
func (w *resultDeliveryWriter) Write(content []byte) (int, error) {
|
||||||
|
if w.err != nil {
|
||||||
|
return 0, w.err
|
||||||
|
}
|
||||||
|
return w.accepted.Write(content)
|
||||||
|
}
|
||||||
200
internal/cli/run_result_test.go
Normal file
200
internal/cli/run_result_test.go
Normal file
@@ -0,0 +1,200 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"encoding/json"
|
||||||
|
"errors"
|
||||||
|
"io"
|
||||||
|
"path/filepath"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestRunResultEncodesRequiredFieldsAndCounts(t *testing.T) {
|
||||||
|
result, err := newRunResult(testResolvedPipeline(pipeline.DefaultOutputModule), testRunOutput(), "relative-output", "relative-debug")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
encoded, err := encodeRunResult(result)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if encoded[len(encoded)-1] != '\n' || bytes.Count(encoded, []byte{'\n'}) != 1 {
|
||||||
|
t.Fatalf("encoded result is not one newline-terminated object: %q", encoded)
|
||||||
|
}
|
||||||
|
|
||||||
|
var decoded map[string]any
|
||||||
|
if err := json.Unmarshal(encoded, &decoded); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if got := decoded["schema_version"]; got != runResultSchemaVersion {
|
||||||
|
t.Fatalf("schema_version = %q", got)
|
||||||
|
}
|
||||||
|
if got := decoded["run_id"]; got != "run-123" {
|
||||||
|
t.Fatalf("run_id = %q", got)
|
||||||
|
}
|
||||||
|
if got := decoded["pipeline_id"]; got != "sample" {
|
||||||
|
t.Fatalf("pipeline_id = %q", got)
|
||||||
|
}
|
||||||
|
if got := decoded["validation_status"]; got != "rejected" {
|
||||||
|
t.Fatalf("validation_status = %q", got)
|
||||||
|
}
|
||||||
|
if got := decoded["index_file"]; got != "index.json" {
|
||||||
|
t.Fatalf("index_file = %q", got)
|
||||||
|
}
|
||||||
|
if got := decoded["normalized_output_count"]; got != float64(2) {
|
||||||
|
t.Fatalf("normalized_output_count = %v", got)
|
||||||
|
}
|
||||||
|
if got := decoded["rejected_output_count"]; got != float64(1) {
|
||||||
|
t.Fatalf("rejected_output_count = %v", got)
|
||||||
|
}
|
||||||
|
if got := decoded["warning_count"]; got != float64(1) {
|
||||||
|
t.Fatalf("warning_count = %v", got)
|
||||||
|
}
|
||||||
|
if got := decoded["output_directory"]; got != filepath.Join(mustWorkingDirectory(t), "relative-output") {
|
||||||
|
t.Fatalf("output_directory = %q", got)
|
||||||
|
}
|
||||||
|
if got := decoded["debug_directory"]; got != filepath.Join(mustWorkingDirectory(t), "relative-debug") {
|
||||||
|
t.Fatalf("debug_directory = %q", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunResultRejectsInvalidRequiredValues(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
resolved pipeline.ResolvedPipeline
|
||||||
|
output pipeline.RunOutput
|
||||||
|
directory string
|
||||||
|
}{
|
||||||
|
{name: "blank run ID", resolved: testResolvedPipeline(pipeline.DefaultOutputModule), output: testRunOutputWithout(func(output *pipeline.RunOutput) { output.Manifest.RunID = " " }), directory: "output"},
|
||||||
|
{name: "blank resolved pipeline ID", resolved: pipeline.ResolvedPipeline{Output: pipeline.ModuleBinding{Module: pipeline.DefaultOutputModule}}, output: testRunOutput(), directory: "output"},
|
||||||
|
{name: "blank manifest pipeline ID", resolved: testResolvedPipeline(pipeline.DefaultOutputModule), output: testRunOutputWithout(func(output *pipeline.RunOutput) { output.Manifest.PipelineID = "" }), directory: "output"},
|
||||||
|
{name: "mismatched pipeline IDs", resolved: testResolvedPipeline(pipeline.DefaultOutputModule), output: testRunOutputWithout(func(output *pipeline.RunOutput) { output.Manifest.PipelineID = "other" }), directory: "output"},
|
||||||
|
{name: "blank validation status", resolved: testResolvedPipeline(pipeline.DefaultOutputModule), output: testRunOutputWithout(func(output *pipeline.RunOutput) { output.Manifest.ValidationStatus = " " }), directory: "output"},
|
||||||
|
{name: "blank output directory", resolved: testResolvedPipeline(pipeline.DefaultOutputModule), output: testRunOutput(), directory: " "},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
if _, err := newRunResult(tt.resolved, tt.output, tt.directory, ""); err == nil {
|
||||||
|
t.Fatal("newRunResult() succeeded")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunResultOmitsIndexFileForOtherOutputModules(t *testing.T) {
|
||||||
|
result, err := newRunResult(testResolvedPipeline("test/output"), testRunOutputWithout(func(output *pipeline.RunOutput) {
|
||||||
|
output.OutputFiles = nil
|
||||||
|
}), "output", "")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if result.IndexFile != "" {
|
||||||
|
t.Fatalf("index_file = %q", result.IndexFile)
|
||||||
|
}
|
||||||
|
encoded, err := encodeRunResult(result)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var decoded map[string]any
|
||||||
|
if err := json.Unmarshal(encoded, &decoded); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if _, ok := decoded["index_file"]; ok {
|
||||||
|
t.Fatalf("encoded non-JSON result contains index_file: %s", encoded)
|
||||||
|
}
|
||||||
|
if _, ok := decoded["debug_directory"]; ok {
|
||||||
|
t.Fatalf("encoded result without debug capture contains debug_directory: %s", encoded)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunResultRequiresOneProductionIndexFile(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
files []contracts.OutputFile
|
||||||
|
}{
|
||||||
|
{name: "missing", files: nil},
|
||||||
|
{name: "duplicate", files: []contracts.OutputFile{{Name: "index.json"}, {Name: "index.json"}}},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
output := testRunOutput()
|
||||||
|
output.OutputFiles = tt.files
|
||||||
|
if _, err := newRunResult(testResolvedPipeline(pipeline.DefaultOutputModule), output, "output", ""); err == nil {
|
||||||
|
t.Fatal("newRunResult() succeeded")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestWriteRunResultCompletesAndReportsWriterFailure(t *testing.T) {
|
||||||
|
content := []byte("result\n")
|
||||||
|
var target bytes.Buffer
|
||||||
|
if err := writeRunResult(partialResultWriter{writer: &target, limit: 2}, content); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if got := target.String(); got != string(content) {
|
||||||
|
t.Fatalf("written result = %q", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
writerErr := errors.New("result writer failed")
|
||||||
|
if err := writeRunResult(failingResultWriter{err: writerErr}, content); !errors.Is(err, writerErr) {
|
||||||
|
t.Fatalf("writeRunResult() error = %v", err)
|
||||||
|
}
|
||||||
|
if err := writeRunResult(zeroResultWriter{}, content); !errors.Is(err, io.ErrShortWrite) {
|
||||||
|
t.Fatalf("zero-progress error = %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func testResolvedPipeline(outputModule string) pipeline.ResolvedPipeline {
|
||||||
|
return pipeline.ResolvedPipeline{ID: "sample", Output: pipeline.ModuleBinding{Module: outputModule}}
|
||||||
|
}
|
||||||
|
|
||||||
|
func testRunOutput() pipeline.RunOutput {
|
||||||
|
return pipeline.RunOutput{
|
||||||
|
Manifest: artifacts.RunManifest{RunID: "run-123", PipelineID: "sample", ValidationStatus: "rejected"},
|
||||||
|
NormalizeOutputs: []contracts.SerializedOutput{{}, {}},
|
||||||
|
Rejected: []contracts.RejectedOutput{{}},
|
||||||
|
Warnings: []contracts.Warning{{}},
|
||||||
|
OutputFiles: []contracts.OutputFile{{Name: "index.json"}},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func testRunOutputWithout(change func(*pipeline.RunOutput)) pipeline.RunOutput {
|
||||||
|
output := testRunOutput()
|
||||||
|
change(&output)
|
||||||
|
return output
|
||||||
|
}
|
||||||
|
|
||||||
|
func mustWorkingDirectory(t *testing.T) string {
|
||||||
|
t.Helper()
|
||||||
|
workingDirectory, err := filepath.Abs(".")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return workingDirectory
|
||||||
|
}
|
||||||
|
|
||||||
|
type partialResultWriter struct {
|
||||||
|
writer io.Writer
|
||||||
|
limit int
|
||||||
|
}
|
||||||
|
|
||||||
|
func (w partialResultWriter) Write(content []byte) (int, error) {
|
||||||
|
if len(content) > w.limit {
|
||||||
|
content = content[:w.limit]
|
||||||
|
}
|
||||||
|
return w.writer.Write(content)
|
||||||
|
}
|
||||||
|
|
||||||
|
type failingResultWriter struct{ err error }
|
||||||
|
|
||||||
|
func (w failingResultWriter) Write([]byte) (int, error) { return 0, w.err }
|
||||||
|
|
||||||
|
type zeroResultWriter struct{}
|
||||||
|
|
||||||
|
func (zeroResultWriter) Write([]byte) (int, error) { return 0, nil }
|
||||||
90
internal/cli/run_terminal.go
Normal file
90
internal/cli/run_terminal.go
Normal file
@@ -0,0 +1,90 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
|
"io"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/debugbundle"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
)
|
||||||
|
|
||||||
|
type DebugTerminalWriter interface {
|
||||||
|
WriteRunReport(debugbundle.RunReport) error
|
||||||
|
WriteError(string) error
|
||||||
|
}
|
||||||
|
|
||||||
|
type pipelineCommandState struct {
|
||||||
|
report debugbundle.RunReport
|
||||||
|
terminalized bool
|
||||||
|
}
|
||||||
|
|
||||||
|
func newPipelineCommandState(runID, pipelineID, outputPath string) *pipelineCommandState {
|
||||||
|
return &pipelineCommandState{report: debugbundle.RunReport{
|
||||||
|
RunID: runID,
|
||||||
|
PipelineID: pipelineID,
|
||||||
|
OutputPath: outputPath,
|
||||||
|
}}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s *pipelineCommandState) setDebugPath(debugPath string) {
|
||||||
|
if s != nil {
|
||||||
|
s.report.DebugPath = debugPath
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s *pipelineCommandState) observeOutput(output pipeline.RunOutput) {
|
||||||
|
if s == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
s.report.OutputCount = len(output.NormalizeOutputs)
|
||||||
|
s.report.RejectedCount = len(output.Rejected)
|
||||||
|
s.report.WarningCount = len(output.Warnings)
|
||||||
|
s.report.ValidationStatus = output.Manifest.ValidationStatus
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s *pipelineCommandState) terminalize(writer DebugTerminalWriter, primaryErr error) (error, error) {
|
||||||
|
if s == nil || s.terminalized {
|
||||||
|
return primaryErr, nil
|
||||||
|
}
|
||||||
|
s.terminalized = true
|
||||||
|
if writer == nil {
|
||||||
|
return primaryErr, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
report := s.report
|
||||||
|
report.Succeeded = primaryErr == nil
|
||||||
|
reportErr := writer.WriteRunReport(report)
|
||||||
|
if reportErr != nil {
|
||||||
|
reportErr = fmt.Errorf("write debug run report: %w", reportErr)
|
||||||
|
if primaryErr == nil {
|
||||||
|
primaryErr = reportErr
|
||||||
|
reportErr = nil
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
var errorLogErr error
|
||||||
|
if primaryErr != nil {
|
||||||
|
if err := writer.WriteError(primaryErr.Error()); err != nil {
|
||||||
|
errorLogErr = fmt.Errorf("write debug error log: %w", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return primaryErr, errors.Join(reportErr, errorLogErr)
|
||||||
|
}
|
||||||
|
|
||||||
|
func failPipelineCommand(stderr io.Writer, state *pipelineCommandState, writer DebugTerminalWriter, primaryErr error, persistenceErrs ...error) int {
|
||||||
|
primaryErr, terminalErr := state.terminalize(writer, primaryErr)
|
||||||
|
persistenceErrs = append(persistenceErrs, terminalErr)
|
||||||
|
return writePipelineCommandFailure(stderr, state, primaryErr, errors.Join(persistenceErrs...))
|
||||||
|
}
|
||||||
|
|
||||||
|
func writePipelineCommandFailure(stderr io.Writer, state *pipelineCommandState, primaryErr, persistenceErr error) int {
|
||||||
|
fmt.Fprintf(stderr, "notarius: %v\n", primaryErr)
|
||||||
|
if persistenceErr != nil {
|
||||||
|
fmt.Fprintf(stderr, "notarius: %v\n", persistenceErr)
|
||||||
|
}
|
||||||
|
if state != nil && state.report.DebugPath != "" {
|
||||||
|
fmt.Fprintf(stderr, "notarius: debug=%s\n", state.report.DebugPath)
|
||||||
|
}
|
||||||
|
return 1
|
||||||
|
}
|
||||||
File diff suppressed because it is too large
Load Diff
68
internal/cli/scriptorium_profiles.go
Normal file
68
internal/cli/scriptorium_profiles.go
Normal file
@@ -0,0 +1,68 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
|
"testing/fstest"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||||
|
"gitea.maximumdirect.net/eric/scriptorium"
|
||||||
|
)
|
||||||
|
|
||||||
|
const profileCheckPromptID = "notarius.profile.check"
|
||||||
|
|
||||||
|
var profileCheckPromptFS = fstest.MapFS{
|
||||||
|
"prompts/profile-check.yaml": &fstest.MapFile{Data: []byte(`id: notarius.profile.check
|
||||||
|
version: "1.0.0"
|
||||||
|
default_profile: mistral-small-3
|
||||||
|
inputs:
|
||||||
|
- name: transcript
|
||||||
|
required: true
|
||||||
|
messages:
|
||||||
|
- role: user
|
||||||
|
content: "{{input \"transcript\"}}"
|
||||||
|
output:
|
||||||
|
format: text
|
||||||
|
validation_mode: none
|
||||||
|
repair_attempts: 0
|
||||||
|
`)},
|
||||||
|
}
|
||||||
|
|
||||||
|
func validateExplicitScriptoriumProfiles(ctx context.Context, cfg config.Config, profileIDs []string) error {
|
||||||
|
if len(profileIDs) == 0 {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
engine, err := newProfileValidationEngine(cfg)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("load Scriptorium profiles: %w", err)
|
||||||
|
}
|
||||||
|
for _, profileID := range profileIDs {
|
||||||
|
if _, err := engine.Prepare(ctx, scriptorium.RunRequest{
|
||||||
|
PromptID: profileCheckPromptID,
|
||||||
|
ProfileID: profileID,
|
||||||
|
Inputs: map[string]scriptorium.ArtifactRef{
|
||||||
|
"transcript": scriptorium.Inline("profile check"),
|
||||||
|
},
|
||||||
|
}); err != nil {
|
||||||
|
if errors.Is(err, scriptorium.ErrProfileNotFound) {
|
||||||
|
return fmt.Errorf("Scriptorium profile %q is not configured", profileID)
|
||||||
|
}
|
||||||
|
return fmt.Errorf("validate Scriptorium profile %q: %w", profileID, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func newProfileValidationEngine(cfg config.Config) (*scriptorium.Engine, error) {
|
||||||
|
opts := []scriptorium.Option{
|
||||||
|
scriptorium.WithPromptFS(profileCheckPromptFS, "prompts"),
|
||||||
|
}
|
||||||
|
if cfg.Scriptorium.ProfileFile != "" {
|
||||||
|
opts = append(opts, scriptorium.WithProfileFile(cfg.Scriptorium.ProfileFile))
|
||||||
|
}
|
||||||
|
return scriptorium.NewEngine(scriptorium.Config{
|
||||||
|
PromptDir: "unused",
|
||||||
|
ProfileDir: cfg.Scriptorium.ProfileDir,
|
||||||
|
}, opts...)
|
||||||
|
}
|
||||||
490
internal/cli/spell_catalog_identity_contract_test.go
Normal file
490
internal/cli/spell_catalog_identity_contract_test.go
Normal file
@@ -0,0 +1,490 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"crypto/sha256"
|
||||||
|
"encoding/hex"
|
||||||
|
"encoding/json"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"reflect"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/checkpoint"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/spells"
|
||||||
|
spellnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/spells"
|
||||||
|
spellcatalog "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/spells/catalog"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestSpellCatalogBytesAffectCheckpointIdentityButNotSemanticDigest(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
configPath := writeProductionSpellCatalogContractConfig(t)
|
||||||
|
effective, err := loadMaintainedExample(t, configPath).Resolve(resolveInputForMaintainedExample(components, "dnd-session"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("resolve production configuration: %v", err)
|
||||||
|
}
|
||||||
|
overlayPath := filepath.Join(t.TempDir(), "catalog.json")
|
||||||
|
resolved := effective.ResolvedPipeline
|
||||||
|
bindings := resolved.Steps[0].ArtifactLanes[0].ExtractReferences.Bindings
|
||||||
|
catalogBindingIndex := -1
|
||||||
|
for index, binding := range bindings {
|
||||||
|
if binding.SlotName == spellcatalog.SpellCatalogReferenceSlot {
|
||||||
|
catalogBindingIndex = index
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if catalogBindingIndex < 0 {
|
||||||
|
t.Fatalf("spell catalog bindings = %#v, want catalog binding", bindings)
|
||||||
|
}
|
||||||
|
resolved.Steps[0].ArtifactLanes[0].ExtractReferences.Bindings[catalogBindingIndex].Source = overlayPath
|
||||||
|
normalizeBindings := resolved.Steps[0].ArtifactLanes[0].NormalizeReferences.Bindings
|
||||||
|
normalizeCatalogBindingIndex := -1
|
||||||
|
for index, binding := range normalizeBindings {
|
||||||
|
if binding.SlotName == spellcatalog.SpellCatalogReferenceSlot {
|
||||||
|
normalizeCatalogBindingIndex = index
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if normalizeCatalogBindingIndex < 0 {
|
||||||
|
t.Fatalf("normalize spell catalog bindings = %#v, want catalog binding", normalizeBindings)
|
||||||
|
}
|
||||||
|
resolved.Steps[0].ArtifactLanes[0].NormalizeReferences.Bindings[normalizeCatalogBindingIndex].Source = overlayPath
|
||||||
|
|
||||||
|
if err := os.WriteFile(overlayPath, []byte(reorderedOverlayA), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
materializedA, _, err := pipeline.MaterializeReferences(resolved, catalogFromRegistries(components.registries), pipeline.ReferenceMaterializationOptions{ConfigPath: configPath, WorkingDir: filepath.Dir(configPath)})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("materialize first catalog: %v", err)
|
||||||
|
}
|
||||||
|
identityA := catalogCheckpointIdentity(t, materializedA)
|
||||||
|
metadataA := catalogExtractorMetadata(t, materializedA)
|
||||||
|
normalizerMetadataA := catalogNormalizerMetadata(t, materializedA)
|
||||||
|
referenceA := catalogReference(t, materializedA)
|
||||||
|
|
||||||
|
if err := os.WriteFile(overlayPath, []byte(reorderedOverlayB), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
materializedB, _, err := pipeline.MaterializeReferences(resolved, catalogFromRegistries(components.registries), pipeline.ReferenceMaterializationOptions{ConfigPath: configPath, WorkingDir: filepath.Dir(configPath)})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("materialize reordered catalog: %v", err)
|
||||||
|
}
|
||||||
|
identityB := catalogCheckpointIdentity(t, materializedB)
|
||||||
|
metadataB := catalogExtractorMetadata(t, materializedB)
|
||||||
|
normalizerMetadataB := catalogNormalizerMetadata(t, materializedB)
|
||||||
|
referenceB := catalogReference(t, materializedB)
|
||||||
|
|
||||||
|
if identityA.Digest == identityB.Digest {
|
||||||
|
t.Fatalf("checkpoint identity digest = %q for both raw catalog files, want invalidation", identityA.Digest)
|
||||||
|
}
|
||||||
|
if referenceA.Digest == referenceB.Digest || referenceA.OriginURI != referenceB.OriginURI {
|
||||||
|
t.Fatalf("catalog reference provenance changed from %#v to %#v, want same origin and different raw digest", referenceA, referenceB)
|
||||||
|
}
|
||||||
|
digestA, ok := metadataA["catalog_digest"].(string)
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("first extractor catalog metadata = %#v, want digest", metadataA)
|
||||||
|
}
|
||||||
|
digestB, ok := metadataB["catalog_digest"].(string)
|
||||||
|
if !ok || digestA != digestB {
|
||||||
|
t.Fatalf("extractor catalog digests = %q and %q, want same semantic digest", digestA, digestB)
|
||||||
|
}
|
||||||
|
if got, want := metadataA["catalog_overlay_ids"], []string{"campaign.a", "campaign.b"}; !reflect.DeepEqual(got, want) || !reflect.DeepEqual(metadataB["catalog_overlay_ids"], want) {
|
||||||
|
t.Fatalf("extractor overlay IDs = %#v and %#v, want %#v", got, metadataB["catalog_overlay_ids"], want)
|
||||||
|
}
|
||||||
|
normalizerDigestA, ok := normalizerMetadataA["catalog_digest"].(string)
|
||||||
|
normalizerDigestB, okB := normalizerMetadataB["catalog_digest"].(string)
|
||||||
|
if !ok || !okB || normalizerDigestA != digestA || normalizerDigestB != digestB {
|
||||||
|
t.Fatalf("normalizer catalog digests = %#v and %#v, want extractor semantic digests %q and %q", normalizerMetadataA["catalog_digest"], normalizerMetadataB["catalog_digest"], digestA, digestB)
|
||||||
|
}
|
||||||
|
if got, want := normalizerMetadataA["catalog_overlay_ids"], []string{"campaign.a", "campaign.b"}; !reflect.DeepEqual(got, want) || !reflect.DeepEqual(normalizerMetadataB["catalog_overlay_ids"], want) {
|
||||||
|
t.Fatalf("normalizer overlay IDs = %#v and %#v, want %#v", got, normalizerMetadataB["catalog_overlay_ids"], want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestConfiguredSpellCatalogBindingChangesResolvedPipelineIdentity(t *testing.T) {
|
||||||
|
base := productionSpellCatalogContractConfig(t)
|
||||||
|
changed := strings.Replace(base, repositoryPath("examples", "dnd-spell-catalog.json"), filepath.Join(t.TempDir(), "alternate-spell-catalog.json"), 1)
|
||||||
|
if changed == base {
|
||||||
|
t.Fatal("production configuration did not contain the maintained catalog binding")
|
||||||
|
}
|
||||||
|
root := t.TempDir()
|
||||||
|
firstPath := filepath.Join(root, "first.yml")
|
||||||
|
secondPath := filepath.Join(root, "second.yml")
|
||||||
|
if err := os.WriteFile(firstPath, []byte(base), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(secondPath, []byte(changed), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
first, err := loadMaintainedExample(t, firstPath).Resolve(resolveInputForMaintainedExample(components, "dnd-session"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("resolve first configuration: %v", err)
|
||||||
|
}
|
||||||
|
second, err := loadMaintainedExample(t, secondPath).Resolve(resolveInputForMaintainedExample(components, "dnd-session"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("resolve changed configuration: %v", err)
|
||||||
|
}
|
||||||
|
if first.ResolvedPipeline.Digest == second.ResolvedPipeline.Digest {
|
||||||
|
t.Fatalf("resolved pipeline digest = %q for different catalog bindings, want change", first.ResolvedPipeline.Digest)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSemanticSpellCatalogFingerprintChangesCheckpointIdentityWithoutReferenceChange(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
configPath := writeProductionSpellCatalogContractConfig(t)
|
||||||
|
effective, err := loadMaintainedExample(t, configPath).Resolve(resolveInputForMaintainedExample(components, "dnd-session"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
materialized, _, err := pipeline.MaterializeReferences(effective.ResolvedPipeline, catalogFromRegistries(components.registries), pipeline.ReferenceMaterializationOptions{ConfigPath: configPath, WorkingDir: filepath.Dir(configPath)})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
prepared, err := pipeline.Prepare(materialized, components.registries, pipeline.ModuleDependencies{LLM: &productionFakeLLMClient{}})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
fingerprints := prepared.CheckpointFingerprints()
|
||||||
|
wantNames := map[string]struct{}{
|
||||||
|
"extract:spells:" + spells.Key + ":effective_catalog": {},
|
||||||
|
"extract:spells:" + spells.Key + ":validator:2:extract/dnd/spells/catalog:effective_catalog": {},
|
||||||
|
"normalize:spells:" + spellnormalize.Key + ":effective_catalog": {},
|
||||||
|
"normalize:spells:" + spellnormalize.Key + ":validator:2:extract/dnd/spells/catalog:effective_catalog": {},
|
||||||
|
}
|
||||||
|
seen := make(map[string]string, len(fingerprints))
|
||||||
|
for _, fingerprint := range fingerprints {
|
||||||
|
if _, ok := wantNames[fingerprint.Name]; ok {
|
||||||
|
seen[fingerprint.Name] = fingerprint.Value
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(seen) != len(wantNames) {
|
||||||
|
t.Fatalf("prepared fingerprints = %#v, want scoped extractor and normalize catalog identities", fingerprints)
|
||||||
|
}
|
||||||
|
var catalogDigest string
|
||||||
|
for name, value := range seen {
|
||||||
|
if catalogDigest == "" {
|
||||||
|
catalogDigest = value
|
||||||
|
} else if value != catalogDigest {
|
||||||
|
t.Fatalf("prepared fingerprint %q = %q, want shared semantic catalog digest %q", name, value, catalogDigest)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
identityFor := func(values []pipeline.CheckpointFingerprint) checkpoint.Identity {
|
||||||
|
identity, identityErr := checkpoint.NewIdentity(checkpoint.IdentityInput{
|
||||||
|
Pipeline: materialized,
|
||||||
|
InputKey: materialized.Input.Module,
|
||||||
|
RawInputDigest: "sha256:unchanged-input",
|
||||||
|
References: pipeline.ReferenceProvenance(materialized),
|
||||||
|
ProvenanceFingerprints: checkpointIdentityFingerprints(values),
|
||||||
|
})
|
||||||
|
if identityErr != nil {
|
||||||
|
t.Fatal(identityErr)
|
||||||
|
}
|
||||||
|
return identity
|
||||||
|
}
|
||||||
|
first := identityFor(fingerprints)
|
||||||
|
changed := replaceCheckpointFingerprintValue(t, fingerprints, normalizeSpellCatalogFingerprintName(), "sha256:changed-effective-catalog")
|
||||||
|
assertOnlyCheckpointFingerprintChanged(t, fingerprints, changed, normalizeSpellCatalogFingerprintName())
|
||||||
|
second := identityFor(changed)
|
||||||
|
if first.Digest == second.Digest || reflect.DeepEqual(first.ReferenceDigests, nil) || !reflect.DeepEqual(first.ReferenceDigests, second.ReferenceDigests) {
|
||||||
|
t.Fatalf("identities = %#v / %#v, want semantic invalidation with unchanged reference provenance", first, second)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestChangedSemanticSpellCatalogFingerprintCannotResumeRecordedCheckpoint(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
configPath := writeProductionSpellCatalogContractConfig(t)
|
||||||
|
effective, err := loadMaintainedExample(t, configPath).Resolve(resolveInputForMaintainedExample(components, "dnd-session"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
materialized, _, err := pipeline.MaterializeReferences(effective.ResolvedPipeline, catalogFromRegistries(components.registries), pipeline.ReferenceMaterializationOptions{ConfigPath: configPath, WorkingDir: filepath.Dir(configPath)})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
prepared, err := pipeline.Prepare(materialized, components.registries, pipeline.ModuleDependencies{LLM: &productionFakeLLMClient{}})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
fingerprints := prepared.CheckpointFingerprints()
|
||||||
|
settings := config.CheckpointCacheConfig{Enabled: true, Directory: t.TempDir()}
|
||||||
|
recorder, _, err := checkpointHandlersForRun(settings, Options{}, materialized, fingerprints, []byte("same input"), nil, nil, "", "", false)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
doc := source.SourceDocument{ID: "source", Kind: "transcript", Format: "application/json"}
|
||||||
|
doc.Units = []source.SourceUnit{{ID: 1, Kind: "turn", Text: "Aria casts Cure Wounds.", Ref: source.SourceRef{SourceID: doc.ID, StartUnitID: 1, EndUnitID: 1}}}
|
||||||
|
doc.Digest, err = source.DigestDocument(&doc)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := recorder.SourceSucceeded(materialized.Input.Module, &doc); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
normalizeDependencies := []pipeline.CheckpointFingerprint{{Name: "artifact[0]", Value: "sha256:merged-artifact"}}
|
||||||
|
normalizeSchema := contracts.ArtifactSchema{ID: "notarius.dnd.spells", Name: "notarius_dnd_spells", Version: "v1"}
|
||||||
|
normalizeArtifact := pipeline.CheckpointArtifact{
|
||||||
|
LaneID: "spells", ModuleKey: spellnormalize.Key, SourceID: doc.ID,
|
||||||
|
SchemaDigest: contracts.DigestArtifactSchema(normalizeSchema),
|
||||||
|
Artifact: contracts.SerializedArtifact{
|
||||||
|
Kind: dnd.SpellListKind, Schema: normalizeSchema, MediaType: "application/json", Content: []byte(`{"spell_casts":[]}`),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
if err := recorder.NormalizeSucceeded("spells", spellnormalize.Key, normalizeDependencies, normalizeArtifact, nil); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
_, sameLoader, err := checkpointHandlersForRun(settings, Options{}, materialized, fingerprints, []byte("same input"), nil, nil, "", "", true)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if _, decision := sameLoader.Source(materialized.Input.Module); !decision.Reused {
|
||||||
|
t.Fatalf("same fingerprint decision = %#v, want reuse", decision)
|
||||||
|
}
|
||||||
|
if restored, decision := sameLoader.Normalize("spells", spellnormalize.Key, normalizeDependencies); !decision.Reused || string(restored.Output.Artifact.Content) != `{"spell_casts":[]}` {
|
||||||
|
t.Fatalf("same normalize checkpoint = %#v, decision=%#v, want reuse", restored, decision)
|
||||||
|
}
|
||||||
|
changed := replaceCheckpointFingerprintValue(t, fingerprints, normalizeSpellCatalogFingerprintName(), "sha256:changed-effective-catalog")
|
||||||
|
assertOnlyCheckpointFingerprintChanged(t, fingerprints, changed, normalizeSpellCatalogFingerprintName())
|
||||||
|
_, changedLoader, err := checkpointHandlersForRun(settings, Options{}, materialized, changed, []byte("same input"), nil, nil, "", "", true)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if _, decision := changedLoader.Source(materialized.Input.Module); decision.Reused {
|
||||||
|
t.Fatalf("changed fingerprint decision = %#v, want cold miss", decision)
|
||||||
|
}
|
||||||
|
if _, decision := changedLoader.Normalize("spells", spellnormalize.Key, normalizeDependencies); decision.Reused {
|
||||||
|
t.Fatalf("changed normalize fingerprint decision = %#v, want normalize checkpoint cold miss", decision)
|
||||||
|
}
|
||||||
|
changedMapping := replaceCheckpointFingerprintValue(t, fingerprints, extractSpellMappingFingerprintName(), "dnd.spells.extract_mapping.v3")
|
||||||
|
assertOnlyCheckpointFingerprintChanged(t, fingerprints, changedMapping, extractSpellMappingFingerprintName())
|
||||||
|
_, mappingLoader, err := checkpointHandlersForRun(settings, Options{}, materialized, changedMapping, []byte("same input"), nil, nil, "", "", true)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if _, decision := mappingLoader.Source(materialized.Input.Module); decision.Reused {
|
||||||
|
t.Fatalf("changed mapping policy decision = %#v, want cold miss", decision)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func normalizeSpellCatalogFingerprintName() string {
|
||||||
|
return "normalize:spells:" + spellnormalize.Key + ":effective_catalog"
|
||||||
|
}
|
||||||
|
|
||||||
|
func extractSpellMappingFingerprintName() string {
|
||||||
|
return "extract:spells:" + spells.Key + ":mapping_policy"
|
||||||
|
}
|
||||||
|
|
||||||
|
func replaceCheckpointFingerprintValue(t *testing.T, fingerprints []pipeline.CheckpointFingerprint, name, value string) []pipeline.CheckpointFingerprint {
|
||||||
|
t.Helper()
|
||||||
|
changed := append([]pipeline.CheckpointFingerprint(nil), fingerprints...)
|
||||||
|
matches := 0
|
||||||
|
for index := range changed {
|
||||||
|
if changed[index].Name == name {
|
||||||
|
changed[index].Value = value
|
||||||
|
matches++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if matches != 1 {
|
||||||
|
t.Fatalf("checkpoint fingerprints = %#v, want exactly one fingerprint named %q", fingerprints, name)
|
||||||
|
}
|
||||||
|
return changed
|
||||||
|
}
|
||||||
|
|
||||||
|
func assertOnlyCheckpointFingerprintChanged(t *testing.T, before, after []pipeline.CheckpointFingerprint, changedName string) {
|
||||||
|
t.Helper()
|
||||||
|
if len(before) != len(after) {
|
||||||
|
t.Fatalf("fingerprint lengths = %d and %d, want equal", len(before), len(after))
|
||||||
|
}
|
||||||
|
changes := 0
|
||||||
|
for index := range before {
|
||||||
|
if before[index].Name != after[index].Name {
|
||||||
|
t.Fatalf("fingerprint[%d] name changed from %q to %q", index, before[index].Name, after[index].Name)
|
||||||
|
}
|
||||||
|
if before[index].Value == after[index].Value {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
changes++
|
||||||
|
if before[index].Name != changedName {
|
||||||
|
t.Fatalf("fingerprint %q changed unexpectedly", before[index].Name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if changes != 1 {
|
||||||
|
t.Fatalf("fingerprints changed %d values, want exactly %q", changes, changedName)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestMaintainedProductionOverlayRunAlignsGroundingValidationAndProvenance(t *testing.T) {
|
||||||
|
outputRoot := filepath.Join(t.TempDir(), "output")
|
||||||
|
fake := &productionFakeLLMClient{spellResponse: productionSpellResponse("Aegis of Emberfall")}
|
||||||
|
options := productionRunOptions(t, fake)
|
||||||
|
var stdout, stderr strings.Builder
|
||||||
|
code := RunWithOptions([]string{
|
||||||
|
"run", "dnd-session",
|
||||||
|
"--config", writeProductionSpellCatalogContractConfig(t),
|
||||||
|
"--input", repositoryPath("examples", "seriatim-minimal-transcript.json"),
|
||||||
|
"--only", "spells", "--chunk_cache", "bypass", "--output-dir", outputRoot,
|
||||||
|
}, &stdout, &stderr, options)
|
||||||
|
if code != 0 {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
runRoot := filepath.Join(outputRoot, productionRunID)
|
||||||
|
manifest := readProductionJSON[artifacts.RunManifest](t, filepath.Join(runRoot, "manifest.json"))
|
||||||
|
if manifest.ValidationStatus != "approved" || len(manifest.References) == 0 || len(manifest.ArtifactLanes) != 1 {
|
||||||
|
t.Fatalf("manifest = %#v, want approved overlay run with one lane and references", manifest)
|
||||||
|
}
|
||||||
|
lane := manifest.ArtifactLanes[0]
|
||||||
|
extractorMetadata, ok := lane.Metadata["extractor"].(map[string]any)
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("lane metadata = %#v, want extractor metadata", lane.Metadata)
|
||||||
|
}
|
||||||
|
if extractorMetadata["catalog_base_id"] != spellcatalog.SRD5E2014ID || !strings.HasPrefix(stringValue(extractorMetadata["catalog_digest"]), "sha256:") {
|
||||||
|
t.Fatalf("extractor catalog metadata = %#v, want base ID and semantic digest", extractorMetadata)
|
||||||
|
}
|
||||||
|
if got := stringValues(extractorMetadata["catalog_overlay_ids"]); !reflect.DeepEqual(got, []string{"notarius.example-campaign"}) {
|
||||||
|
t.Fatalf("catalog overlay IDs = %#v, want maintained overlay", got)
|
||||||
|
}
|
||||||
|
normalizerMetadata, ok := lane.Metadata["normalizer"].(map[string]any)
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("lane metadata = %#v, want normalizer metadata", lane.Metadata)
|
||||||
|
}
|
||||||
|
if normalizerMetadata["catalog_base_id"] != spellcatalog.SRD5E2014ID || !strings.HasPrefix(stringValue(normalizerMetadata["catalog_digest"]), "sha256:") || !reflect.DeepEqual(stringValues(normalizerMetadata["catalog_overlay_ids"]), []string{"notarius.example-campaign"}) {
|
||||||
|
t.Fatalf("normalizer catalog metadata = %#v, want base ID, semantic digest, and overlay IDs", normalizerMetadata)
|
||||||
|
}
|
||||||
|
if normalizerMetadata["catalog_digest"] != extractorMetadata["catalog_digest"] || !reflect.DeepEqual(stringValues(normalizerMetadata["catalog_overlay_ids"]), stringValues(extractorMetadata["catalog_overlay_ids"])) {
|
||||||
|
t.Fatalf("extractor metadata = %#v, normalizer metadata = %#v, want shared catalog identity", extractorMetadata, normalizerMetadata)
|
||||||
|
}
|
||||||
|
|
||||||
|
var catalogProvenances []artifacts.ReferenceProvenance
|
||||||
|
for index := range manifest.References {
|
||||||
|
reference := &manifest.References[index]
|
||||||
|
if reference.SlotName == spellcatalog.SpellCatalogReferenceSlot {
|
||||||
|
catalogProvenances = append(catalogProvenances, *reference)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(catalogProvenances) != 2 {
|
||||||
|
t.Fatalf("manifest references = %#v, want independently materialized extract and normalize catalog provenance", manifest.References)
|
||||||
|
}
|
||||||
|
overlayBytes := readRepositoryFile(t, "examples", "dnd-spell-catalog.json")
|
||||||
|
for _, catalogProvenance := range catalogProvenances {
|
||||||
|
if catalogProvenance.Stage != "extract" && catalogProvenance.Stage != "normalize" {
|
||||||
|
t.Fatalf("catalog provenance = %#v, want extract or normalize scope", catalogProvenance)
|
||||||
|
}
|
||||||
|
if catalogProvenance.LaneID != "spells" || catalogProvenance.OriginType != "file" || catalogProvenance.MediaType != "application/json" || catalogProvenance.SizeBytes != int64(len(overlayBytes)) || catalogProvenance.Digest != digestBytes(overlayBytes) || !strings.Contains(catalogProvenance.OriginURI, "dnd-spell-catalog.json") {
|
||||||
|
t.Fatalf("catalog provenance = %#v, want raw overlay provenance in both scopes", catalogProvenance)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
manifestBytes, err := json.Marshal(manifest)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
for _, leaked := range []string{"Aegis of Emberfall", "Emberfall Aegis", "Notarius example campaign spell names"} {
|
||||||
|
if strings.Contains(string(manifestBytes), leaked) {
|
||||||
|
t.Fatalf("manifest leaked overlay content %q", leaked)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
requests := fake.requestsFor(spells.PromptID)
|
||||||
|
if len(requests) != 1 {
|
||||||
|
t.Fatalf("spell requests = %d, want one", len(requests))
|
||||||
|
}
|
||||||
|
catalogInput, ok := requests[0].Inputs[spellcatalog.SpellCatalogReferenceSlot]
|
||||||
|
if !ok || !strings.Contains(string(catalogInput.Content), "Aegis of Emberfall") || strings.Contains(string(catalogInput.Content), "Emberfall Aegis") {
|
||||||
|
t.Fatalf("spell catalog prompt input = %#v, want canonical overlay name without alias", catalogInput)
|
||||||
|
}
|
||||||
|
artifact := readProductionJSON[dnd.SpellList](t, filepath.Join(runRoot, "lanes", "spells.json"))
|
||||||
|
if len(artifact.SpellCasts) != 1 || artifact.SpellCasts[0].Spell != "Aegis of Emberfall" {
|
||||||
|
t.Fatalf("artifact = %#v, want accepted overlay-only canonical spell", artifact)
|
||||||
|
}
|
||||||
|
rejected := readProductionJSON[struct {
|
||||||
|
Rejected []json.RawMessage `json:"rejected"`
|
||||||
|
}](t, filepath.Join(runRoot, "rejected.json"))
|
||||||
|
if len(rejected.Rejected) != 0 {
|
||||||
|
t.Fatalf("rejected = %#v, want no rejected output", rejected.Rejected)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func catalogCheckpointIdentity(t *testing.T, resolved pipeline.ResolvedPipeline) checkpoint.Identity {
|
||||||
|
t.Helper()
|
||||||
|
identity, err := checkpoint.NewIdentity(checkpoint.IdentityInput{
|
||||||
|
Pipeline: resolved,
|
||||||
|
InputKey: resolved.Input.Module,
|
||||||
|
RawInputDigest: "sha256:catalog-test-input",
|
||||||
|
References: pipeline.ReferenceProvenance(resolved),
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("create checkpoint identity: %v", err)
|
||||||
|
}
|
||||||
|
return identity
|
||||||
|
}
|
||||||
|
|
||||||
|
func catalogExtractorMetadata(t *testing.T, resolved pipeline.ResolvedPipeline) map[string]any {
|
||||||
|
t.Helper()
|
||||||
|
lane := resolved.Steps[0].ArtifactLanes[0]
|
||||||
|
extractor, err := spells.New(&productionFakeLLMClient{}, spells.Options{}, lane.ExtractReferences.ReferenceSet)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("construct extractor: %v", err)
|
||||||
|
}
|
||||||
|
return extractor.ManifestMetadata()
|
||||||
|
}
|
||||||
|
|
||||||
|
func catalogNormalizerMetadata(t *testing.T, resolved pipeline.ResolvedPipeline) map[string]any {
|
||||||
|
t.Helper()
|
||||||
|
lane := resolved.Steps[0].ArtifactLanes[0]
|
||||||
|
normalizer, err := spellnormalize.New(spellnormalize.Options{}, lane.NormalizeReferences.ReferenceSet)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("construct normalizer: %v", err)
|
||||||
|
}
|
||||||
|
return normalizer.ManifestMetadata()
|
||||||
|
}
|
||||||
|
|
||||||
|
func catalogReference(t *testing.T, resolved pipeline.ResolvedPipeline) artifacts.ReferenceProvenance {
|
||||||
|
t.Helper()
|
||||||
|
for _, reference := range pipeline.ReferenceProvenance(resolved) {
|
||||||
|
if reference.SlotName == spellcatalog.SpellCatalogReferenceSlot && reference.Stage == "extract" && reference.LaneID == "spells" {
|
||||||
|
return reference
|
||||||
|
}
|
||||||
|
}
|
||||||
|
t.Fatalf("resolved references = %#v, want spell catalog provenance", pipeline.ReferenceProvenance(resolved))
|
||||||
|
return artifacts.ReferenceProvenance{}
|
||||||
|
}
|
||||||
|
|
||||||
|
func stringValue(value any) string {
|
||||||
|
result, _ := value.(string)
|
||||||
|
return result
|
||||||
|
}
|
||||||
|
|
||||||
|
func stringValues(value any) []string {
|
||||||
|
raw, err := json.Marshal(value)
|
||||||
|
if err != nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
var values []string
|
||||||
|
if err := json.Unmarshal(raw, &values); err != nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
return values
|
||||||
|
}
|
||||||
|
|
||||||
|
func digestBytes(value []byte) string {
|
||||||
|
sum := sha256.Sum256(value)
|
||||||
|
return "sha256:" + hex.EncodeToString(sum[:])
|
||||||
|
}
|
||||||
|
|
||||||
|
const reorderedOverlayA = `{
|
||||||
|
"schema_version": "notarius.dnd.spell-catalog-overlay.v1",
|
||||||
|
"catalogs": [
|
||||||
|
{"id":"campaign.a","ruleset":"dnd-5e-2014","source":{"title":"Campaign A"},"spells":[{"name":"Aegis of Emberfall","aliases":["Emberfall Aegis"]}]},
|
||||||
|
{"id":"campaign.b","ruleset":"dnd-5e-2014","source":{"title":"Campaign B"},"spells":[{"name":"Cinder Veil","aliases":["Veil of Cinder","Cinder Shroud"]}]}
|
||||||
|
]
|
||||||
|
}`
|
||||||
|
|
||||||
|
const reorderedOverlayB = `{"catalogs":[{"spells":[{"aliases":["Cinder Shroud","Veil of Cinder"],"name":"Cinder Veil"}],"source":{"title":"Campaign B"},"ruleset":"dnd-5e-2014","id":"campaign.b"},{"spells":[{"aliases":["Emberfall Aegis"],"name":"Aegis of Emberfall"}],"source":{"title":"Campaign A"},"ruleset":"dnd-5e-2014","id":"campaign.a"}],"schema_version":"notarius.dnd.spell-catalog-overlay.v1"}`
|
||||||
160
internal/cli/spell_catalog_retry_contract_test.go
Normal file
160
internal/cli/spell_catalog_retry_contract_test.go
Normal file
@@ -0,0 +1,160 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
|
"path/filepath"
|
||||||
|
"sync"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/spells"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestProductionSpellCatalogValidationRetries(t *testing.T) {
|
||||||
|
const retries = 2
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
responses []string
|
||||||
|
wantCalls int
|
||||||
|
wantRejected bool
|
||||||
|
wantSpell string
|
||||||
|
wantWarningCode string
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
name: "unknown spell remains rejected after exhaustion",
|
||||||
|
responses: []string{
|
||||||
|
productionSpellResponse("Unknown Spell"),
|
||||||
|
productionSpellResponse("Unknown Spell"),
|
||||||
|
productionSpellResponse("Unknown Spell"),
|
||||||
|
},
|
||||||
|
wantCalls: retries + 1,
|
||||||
|
wantRejected: true,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "overlay spell becomes valid on retry",
|
||||||
|
responses: []string{
|
||||||
|
productionSpellResponse("Unknown Spell"),
|
||||||
|
productionSpellResponse("Aegis of Emberfall"),
|
||||||
|
},
|
||||||
|
wantCalls: 2,
|
||||||
|
wantSpell: "Aegis of Emberfall",
|
||||||
|
wantWarningCode: "spell_not_near_source",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
components := productionTestComponents(t)
|
||||||
|
configPath := writeProductionSpellCatalogContractConfig(t)
|
||||||
|
cfg := loadMaintainedExample(t, configPath)
|
||||||
|
effective, err := cfg.Resolve(config.ResolveInput{PipelineID: "dnd-session", Catalog: catalogFromRegistries(components.registries)})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("resolve production configuration: %v", err)
|
||||||
|
}
|
||||||
|
materialized, _, err := pipeline.MaterializeReferences(effective.ResolvedPipeline, catalogFromRegistries(components.registries), pipeline.ReferenceMaterializationOptions{
|
||||||
|
ConfigPath: configPath,
|
||||||
|
WorkingDir: filepath.Dir(configPath),
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("materialize production references: %v", err)
|
||||||
|
}
|
||||||
|
materialized.Steps[0].ArtifactLanes[0].Extract.Retries = retries
|
||||||
|
|
||||||
|
llmClient := &catalogRetryLLMClient{responses: tt.responses}
|
||||||
|
prepared, err := pipeline.Prepare(materialized, components.registries, pipeline.ModuleDependencies{LLM: llmClient})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("prepare production pipeline: %v", err)
|
||||||
|
}
|
||||||
|
output, err := pipeline.New().Run(context.Background(), pipeline.RunInput{
|
||||||
|
Prepared: prepared,
|
||||||
|
RawInput: readRepositoryFile(t, "examples", "seriatim-minimal-transcript.json"),
|
||||||
|
ChunkCacheMode: pipeline.ChunkCacheBypass,
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Run() error = %v, want nil", err)
|
||||||
|
}
|
||||||
|
if calls := llmClient.CallCount(); calls > retries+1 || calls != tt.wantCalls {
|
||||||
|
t.Fatalf("LLM calls = %d, want %d and no more than %d", calls, tt.wantCalls, retries+1)
|
||||||
|
}
|
||||||
|
|
||||||
|
if tt.wantRejected {
|
||||||
|
if len(output.Rejected) != 1 || len(output.NormalizeOutputs) != 0 {
|
||||||
|
t.Fatalf("rejected = %#v normalized = %#v, want one nonfatal rejection and no merge output", output.Rejected, output.NormalizeOutputs)
|
||||||
|
}
|
||||||
|
rejection := output.Rejected[0]
|
||||||
|
if rejection.ReasonCode != "unknown_spell" || rejection.AttemptCount != retries+1 {
|
||||||
|
t.Fatalf("rejection = %#v, want exhausted unknown-spell rejection", rejection)
|
||||||
|
}
|
||||||
|
if len(output.Warnings) != 0 {
|
||||||
|
t.Fatalf("warnings = %#v, want no warnings from rejected attempts", output.Warnings)
|
||||||
|
}
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
if len(output.Rejected) != 0 || len(output.NormalizeOutputs) != 1 {
|
||||||
|
t.Fatalf("rejected = %#v normalized = %#v, want only accepted output", output.Rejected, output.NormalizeOutputs)
|
||||||
|
}
|
||||||
|
var value dnd.SpellList
|
||||||
|
if err := json.Unmarshal(output.NormalizeOutputs[0].Artifact.Content, &value); err != nil {
|
||||||
|
t.Fatalf("decode normalized spell list: %v", err)
|
||||||
|
}
|
||||||
|
if len(value.SpellCasts) != 1 || value.SpellCasts[0].Spell != tt.wantSpell {
|
||||||
|
t.Fatalf("normalized spell list = %#v, want accepted overlay spell", value)
|
||||||
|
}
|
||||||
|
if len(output.Warnings) != 2 || output.Warnings[0].ReasonCode != tt.wantWarningCode || output.Warnings[1].ReasonCode != tt.wantWarningCode {
|
||||||
|
t.Fatalf("warnings = %#v, want accepted-attempt warnings from extract and normalize validation", output.Warnings)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
type catalogRetryLLMClient struct {
|
||||||
|
mu sync.Mutex
|
||||||
|
responses []string
|
||||||
|
calls int
|
||||||
|
}
|
||||||
|
|
||||||
|
func (client *catalogRetryLLMClient) CompleteStructured(ctx context.Context, req contracts.StructuredCompletionRequest, out any) (contracts.StructuredCompletionResponse, error) {
|
||||||
|
if err := ctx.Err(); err != nil {
|
||||||
|
return contracts.StructuredCompletionResponse{}, err
|
||||||
|
}
|
||||||
|
if req.PromptID != spells.PromptID {
|
||||||
|
return contracts.StructuredCompletionResponse{}, fmt.Errorf("unexpected prompt %q", req.PromptID)
|
||||||
|
}
|
||||||
|
client.mu.Lock()
|
||||||
|
index := client.calls
|
||||||
|
client.calls++
|
||||||
|
client.mu.Unlock()
|
||||||
|
if index >= len(client.responses) {
|
||||||
|
return contracts.StructuredCompletionResponse{}, fmt.Errorf("missing fake response %d", index)
|
||||||
|
}
|
||||||
|
content := []byte(client.responses[index])
|
||||||
|
if err := json.Unmarshal(content, out); err != nil {
|
||||||
|
return contracts.StructuredCompletionResponse{}, fmt.Errorf("populate fake structured target: %w", err)
|
||||||
|
}
|
||||||
|
return contracts.StructuredCompletionResponse{Content: content, Provider: "test", Model: "deterministic", ProfileID: req.ProfileID}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (client *catalogRetryLLMClient) CallCount() int {
|
||||||
|
client.mu.Lock()
|
||||||
|
defer client.mu.Unlock()
|
||||||
|
return client.calls
|
||||||
|
}
|
||||||
|
|
||||||
|
func productionSpellResponse(name string) string {
|
||||||
|
content, err := json.Marshal(dnd.SpellList{SpellCasts: []dnd.SpellCast{{
|
||||||
|
Caster: "Aria",
|
||||||
|
Spell: name,
|
||||||
|
SourceRefs: []source.SourceRef{{SourceID: "session-alpha", StartUnitID: 1, EndUnitID: 1}},
|
||||||
|
}}})
|
||||||
|
if err != nil {
|
||||||
|
panic(err)
|
||||||
|
}
|
||||||
|
return string(content)
|
||||||
|
}
|
||||||
988
internal/cli/state_hardening_test.go
Normal file
988
internal/cli/state_hardening_test.go
Normal file
@@ -0,0 +1,988 @@
|
|||||||
|
package cli
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"runtime"
|
||||||
|
"strings"
|
||||||
|
"sync"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/debugbundle"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/chunkplan"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||||
|
frameworkdebug "gitea.maximumdirect.net/eric/notarius/internal/framework/debug"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
)
|
||||||
|
|
||||||
|
const stateTestDigest = "sha256:e511d8906649b78eb639b11215fa57a9652a1a64f4aefa3ed68320dbda46f439"
|
||||||
|
|
||||||
|
func TestRunStateSurfaceMatrix(t *testing.T) {
|
||||||
|
for _, debug := range []bool{false, true} {
|
||||||
|
for _, resume := range []bool{false, true} {
|
||||||
|
for _, mode := range []string{"auto", "bypass", "refresh"} {
|
||||||
|
name := fmt.Sprintf("debug=%t/resume=%t/cache=%s", debug, resume, mode)
|
||||||
|
t.Run(name, func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
opts := harness.options()
|
||||||
|
var storeRoots []string
|
||||||
|
opts.ChunkPlanStoreFactory = func(root string) (pipeline.ChunkPlanStore, error) {
|
||||||
|
storeRoots = append(storeRoots, root)
|
||||||
|
return chunkplan.NewFilesystemStore(root)
|
||||||
|
}
|
||||||
|
result := runStateTest(t, roots, opts, debug, resume, mode)
|
||||||
|
if result.code != 0 {
|
||||||
|
t.Fatalf("code=%d stderr=%q", result.code, result.stderr)
|
||||||
|
}
|
||||||
|
assertStateTestOutput(t, roots.output)
|
||||||
|
if mode == "bypass" {
|
||||||
|
assertAbsent(t, roots.plans)
|
||||||
|
if len(storeRoots) != 0 {
|
||||||
|
t.Fatalf("chunk plan store roots = %v, want none", storeRoots)
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
assertFile(t, filepath.Join(roots.plans, strings.TrimPrefix(stateTestDigest, "sha256:"), "plan.json"))
|
||||||
|
if len(storeRoots) != 1 || storeRoots[0] != roots.plans {
|
||||||
|
t.Fatalf("chunk plan store roots = %v, want [%q]", storeRoots, roots.plans)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assertAnyFile(t, roots.checkpoints)
|
||||||
|
assertRestrictedTree(t, roots.checkpoints)
|
||||||
|
if debug {
|
||||||
|
bundle := onlyChildDir(t, roots.debug)
|
||||||
|
assertFile(t, filepath.Join(bundle, "summary", "invocation.json"))
|
||||||
|
assertAnyFile(t, filepath.Join(bundle, "trace"))
|
||||||
|
assertRestrictedTree(t, roots.debug)
|
||||||
|
} else {
|
||||||
|
assertAbsent(t, roots.debug)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunKeepsStateRootsIndependentAndReusesSelectedCheckpointRoot(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
first := runStateTest(t, roots, harness.options(), true, false, "auto")
|
||||||
|
if first.code != 0 {
|
||||||
|
t.Fatalf("first run code=%d stderr=%q", first.code, first.stderr)
|
||||||
|
}
|
||||||
|
planPath := filepath.Join(roots.plans, strings.TrimPrefix(stateTestDigest, "sha256:"), "plan.json")
|
||||||
|
initialPlan, err := os.ReadFile(planPath)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
firstBundle := onlyChildDir(t, roots.debug)
|
||||||
|
|
||||||
|
second := runStateTest(t, roots, harness.options(), false, false, "auto")
|
||||||
|
if second.code != 0 {
|
||||||
|
t.Fatalf("second run code=%d stderr=%q", second.code, second.stderr)
|
||||||
|
}
|
||||||
|
if harness.chunkCalls != 1 {
|
||||||
|
t.Fatalf("chunk calls after debug toggle = %d, want 1", harness.chunkCalls)
|
||||||
|
}
|
||||||
|
if harness.extractCalls != 2 {
|
||||||
|
t.Fatalf("extract calls after two recording-only runs = %d, want 2", harness.extractCalls)
|
||||||
|
}
|
||||||
|
if got, err := os.ReadFile(planPath); err != nil || !bytes.Equal(got, initialPlan) {
|
||||||
|
t.Fatalf("chunk plan changed after debug toggle: %v", err)
|
||||||
|
}
|
||||||
|
if _, err := os.Stat(firstBundle); err != nil {
|
||||||
|
t.Fatalf("initial debug bundle was removed: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
checkpointRoot := roots.checkpoints
|
||||||
|
extractCallsBeforeResume := harness.extractCalls
|
||||||
|
seed := runStateTest(t, roots, harness.options(), false, true, "auto")
|
||||||
|
if seed.code != 0 {
|
||||||
|
t.Fatalf("checkpoint seed code=%d stderr=%q", seed.code, seed.stderr)
|
||||||
|
}
|
||||||
|
if harness.extractCalls != extractCallsBeforeResume {
|
||||||
|
t.Fatalf("extract calls after reusing recording-only checkpoint = %d, want %d", harness.extractCalls, extractCallsBeforeResume)
|
||||||
|
}
|
||||||
|
extractCalls := harness.extractCalls
|
||||||
|
checkpointFiles := readTree(t, checkpointRoot)
|
||||||
|
reused := runStateTest(t, roots, harness.options(), false, true, "auto")
|
||||||
|
if reused.code != 0 {
|
||||||
|
t.Fatalf("checkpoint reuse code=%d stderr=%q", reused.code, reused.stderr)
|
||||||
|
}
|
||||||
|
if harness.extractCalls != extractCalls {
|
||||||
|
t.Fatalf("extract calls after checkpoint reuse = %d, want %d", harness.extractCalls, extractCalls)
|
||||||
|
}
|
||||||
|
if got := readTree(t, checkpointRoot); !sameFiles(got, checkpointFiles) {
|
||||||
|
t.Fatal("reused checkpoint was rewritten")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunRecomputesOnlyAfterExplicitChunkPlanRemoval(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
first := runStateTest(t, roots, harness.options(), true, false, "auto")
|
||||||
|
if first.code != 0 {
|
||||||
|
t.Fatalf("first run code=%d stderr=%q", first.code, first.stderr)
|
||||||
|
}
|
||||||
|
firstOutput := onlyChildDir(t, roots.output)
|
||||||
|
firstBundle := onlyChildDir(t, roots.debug)
|
||||||
|
entry := filepath.Join(roots.plans, strings.TrimPrefix(stateTestDigest, "sha256:"))
|
||||||
|
if err := os.RemoveAll(entry); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
second := runStateTest(t, roots, harness.options(), false, false, "auto")
|
||||||
|
if second.code != 0 {
|
||||||
|
t.Fatalf("second run code=%d stderr=%q", second.code, second.stderr)
|
||||||
|
}
|
||||||
|
if harness.chunkCalls != 2 {
|
||||||
|
t.Fatalf("chunk calls = %d, want 2 after removing exact cache entry", harness.chunkCalls)
|
||||||
|
}
|
||||||
|
assertFile(t, filepath.Join(firstOutput, "result.json"))
|
||||||
|
assertFile(t, filepath.Join(firstBundle, "summary", "run-report.json"))
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunRetainsDebugBundlesAcrossFailures(t *testing.T) {
|
||||||
|
t.Run("configuration failure precedes allocation", func(t *testing.T) {
|
||||||
|
root := filepath.Join(t.TempDir(), "debug")
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"run", "sample", "--config", filepath.Join(t.TempDir(), "missing.yml"), "--input", "missing", "--debug", "--debug-dir", root}, &stdout, &stderr, newStateTestHarness().options())
|
||||||
|
if code != 1 || !strings.Contains(stderr.String(), "config file") {
|
||||||
|
t.Fatalf("code=%d stderr=%q", code, stderr.String())
|
||||||
|
}
|
||||||
|
assertAbsent(t, root)
|
||||||
|
})
|
||||||
|
|
||||||
|
for _, failure := range []struct {
|
||||||
|
name string
|
||||||
|
expected string
|
||||||
|
setup func(*testing.T, stateTestRoots, *stateTestHarness) Options
|
||||||
|
}{
|
||||||
|
{"resolution", "pipeline \"missing\"", func(t *testing.T, roots stateTestRoots, h *stateTestHarness) Options { return h.options() }},
|
||||||
|
{"pipeline", "synthetic extraction failure", func(t *testing.T, roots stateTestRoots, h *stateTestHarness) Options {
|
||||||
|
h.extractErr = errors.New("synthetic extraction failure")
|
||||||
|
return h.options()
|
||||||
|
}},
|
||||||
|
{"output", "create output parent", func(t *testing.T, roots stateTestRoots, h *stateTestHarness) Options {
|
||||||
|
if err := os.WriteFile(roots.output, []byte("not a directory"), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return h.options()
|
||||||
|
}},
|
||||||
|
{"summary", "write debug invocation metadata", func(t *testing.T, roots stateTestRoots, h *stateTestHarness) Options {
|
||||||
|
opts := h.options()
|
||||||
|
opts.DebugRecorderFactory = func(traceRoot string) (pipeline.DebugRecorder, error) {
|
||||||
|
if err := os.RemoveAll(filepath.Join(filepath.Dir(traceRoot), "summary")); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(filepath.Join(filepath.Dir(traceRoot), "summary"), []byte("blocked"), 0o600); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return frameworkdebug.NewFilesystemRecorder(traceRoot)
|
||||||
|
}
|
||||||
|
return opts
|
||||||
|
}},
|
||||||
|
{"trace", "trace unavailable", func(t *testing.T, roots stateTestRoots, h *stateTestHarness) Options {
|
||||||
|
opts := h.options()
|
||||||
|
opts.DebugRecorderFactory = func(string) (pipeline.DebugRecorder, error) { return failingDebugRecorder{}, nil }
|
||||||
|
return opts
|
||||||
|
}},
|
||||||
|
} {
|
||||||
|
t.Run(failure.name, func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
opts := failure.setup(t, roots, harness)
|
||||||
|
failureStderr := ""
|
||||||
|
if failure.name == "resolution" {
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
code := RunWithOptions([]string{"run", "missing", "--config", roots.config, "--input", roots.input, "--debug"}, &stdout, &stderr, opts)
|
||||||
|
if code != 1 {
|
||||||
|
t.Fatalf("code=%d stderr=%q", code, stderr.String())
|
||||||
|
}
|
||||||
|
failureStderr = stderr.String()
|
||||||
|
} else {
|
||||||
|
result := runStateTest(t, roots, opts, true, false, "bypass")
|
||||||
|
if result.code != 1 {
|
||||||
|
t.Fatalf("code=%d stderr=%q", result.code, result.stderr)
|
||||||
|
}
|
||||||
|
failureStderr = result.stderr
|
||||||
|
}
|
||||||
|
if !strings.Contains(failureStderr, failure.expected) || !strings.Contains(failureStderr, "debug=") {
|
||||||
|
t.Fatalf("stderr=%q, want %q and debug path", failureStderr, failure.expected)
|
||||||
|
}
|
||||||
|
bundle := onlyChildDir(t, roots.debug)
|
||||||
|
if !strings.Contains(readAllFiles(t, bundle), "synthetic") && failure.name == "pipeline" {
|
||||||
|
t.Fatal("pipeline failure was not retained in debug bundle")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunDebugArtifactsRedactSecretsButRetainApplicationData(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
t.Setenv("STATE_TEST_UNRELATED_ENV", "HOST_ONLY_SENTINEL")
|
||||||
|
if err := os.WriteFile(filepath.Join(filepath.Dir(roots.input), "unrelated.txt"), []byte("HOST_ONLY_FILE_SENTINEL"), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
result := runStateTest(t, roots, harness.options(), true, false, "bypass")
|
||||||
|
if result.code != 0 {
|
||||||
|
t.Fatalf("code=%d stderr=%q", result.code, result.stderr)
|
||||||
|
}
|
||||||
|
bundle := onlyChildDir(t, roots.debug)
|
||||||
|
summary := readAllFiles(t, filepath.Join(bundle, "summary"))
|
||||||
|
trace := readAllFiles(t, filepath.Join(bundle, "trace"))
|
||||||
|
for _, forbidden := range []string{"sk-secretvalue", "Bearer secretvalue", "HOST_ONLY_SENTINEL", "HOST_ONLY_FILE_SENTINEL"} {
|
||||||
|
if strings.Contains(summary, forbidden) || strings.Contains(trace, forbidden) {
|
||||||
|
t.Fatalf("debug bundle contains %q", forbidden)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if strings.Contains(summary, "application content") {
|
||||||
|
t.Fatal("summary contains raw application input")
|
||||||
|
}
|
||||||
|
if !strings.Contains(trace, "application content") {
|
||||||
|
t.Fatal("trace does not retain expected application input")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunRedactsSensitiveModuleOptionsFromConfigAndPipelineSummaries(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
data, err := os.ReadFile(roots.config)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
configText := replaceRequiredOnce(t, string(data), " input: test/input\n", ` input:
|
||||||
|
module: test/input
|
||||||
|
options:
|
||||||
|
api_key: CONFIG_SUMMARY_SECRET_SENTINEL
|
||||||
|
safe: SAFE_OPTION_SENTINEL
|
||||||
|
nested:
|
||||||
|
- - password: PIPELINE_SUMMARY_SECRET_SENTINEL
|
||||||
|
neighbor: SAFE_NESTED_OPTION_SENTINEL
|
||||||
|
`)
|
||||||
|
if err := os.WriteFile(roots.config, []byte(configText), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
result := runStateTest(t, roots, newStateTestHarness().options(), true, false, "bypass")
|
||||||
|
if result.code != 0 {
|
||||||
|
t.Fatalf("code=%d stderr=%q", result.code, result.stderr)
|
||||||
|
}
|
||||||
|
summaryRoot := filepath.Join(onlyChildDir(t, roots.debug), "summary")
|
||||||
|
for _, name := range []string{"effective-config.json", "resolved-pipeline.json"} {
|
||||||
|
contents, err := os.ReadFile(filepath.Join(summaryRoot, name))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
text := string(contents)
|
||||||
|
for _, secret := range []string{"CONFIG_SUMMARY_SECRET_SENTINEL", "PIPELINE_SUMMARY_SECRET_SENTINEL"} {
|
||||||
|
if strings.Contains(text, secret) {
|
||||||
|
t.Fatalf("%s contains %q: %s", name, secret, text)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, retained := range []string{"[REDACTED]", "SAFE_OPTION_SENTINEL", "SAFE_NESTED_OPTION_SENTINEL"} {
|
||||||
|
if !strings.Contains(text, retained) {
|
||||||
|
t.Fatalf("%s does not contain %q: %s", name, retained, text)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunUsesOneInjectedIdentityForDebugOutputAndManifest(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
opts := harness.options()
|
||||||
|
const runID = "run-1000000000-11111111111111111111111111111111"
|
||||||
|
opts.RunIDGenerator = func(time.Time) (string, error) { return runID, nil }
|
||||||
|
|
||||||
|
result := runStateTest(t, roots, opts, true, false, "bypass")
|
||||||
|
if result.code != 0 {
|
||||||
|
t.Fatalf("code=%d stderr=%q", result.code, result.stderr)
|
||||||
|
}
|
||||||
|
outputPath := filepath.Join(roots.output, runID)
|
||||||
|
debugPath := filepath.Join(roots.debug, runID)
|
||||||
|
assertFile(t, filepath.Join(outputPath, "result.json"))
|
||||||
|
assertFile(t, filepath.Join(debugPath, "summary", "run-manifest.json"))
|
||||||
|
if !strings.Contains(result.stdout, "output="+outputPath) || !strings.Contains(result.stdout, "debug="+debugPath) {
|
||||||
|
t.Fatalf("stdout=%q, want shared run identity", result.stdout)
|
||||||
|
}
|
||||||
|
data, err := os.ReadFile(filepath.Join(debugPath, "summary", "run-manifest.json"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var manifest artifacts.RunManifest
|
||||||
|
if err := json.Unmarshal(data, &manifest); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if manifest.RunID != runID {
|
||||||
|
t.Fatalf("manifest run ID = %q, want %q", manifest.RunID, runID)
|
||||||
|
}
|
||||||
|
wantStartedAt := time.Unix(1, 0).UTC()
|
||||||
|
if manifest.StartedAt == nil || !manifest.StartedAt.Equal(wantStartedAt) {
|
||||||
|
t.Fatalf("manifest started at = %v, want %v", manifest.StartedAt, wantStartedAt)
|
||||||
|
}
|
||||||
|
var invocation debugbundle.Invocation
|
||||||
|
readStateTestSummaryJSON(t, debugPath, "invocation.json", &invocation)
|
||||||
|
if invocation.RunID != runID || !invocation.StartedAt.Equal(wantStartedAt) {
|
||||||
|
t.Fatalf("debug invocation identity = %#v, want run %q at %v", invocation, runID, wantStartedAt)
|
||||||
|
}
|
||||||
|
report := readStateTestRunReport(t, debugPath)
|
||||||
|
if !report.Succeeded || report.RunID != runID || report.PipelineID != "sample" || report.OutputPath != outputPath || report.DebugPath != debugPath || report.OutputCount != 1 || report.RejectedCount != 0 || report.WarningCount != 0 || report.ValidationStatus != "approved" {
|
||||||
|
t.Fatalf("success report = %#v", report)
|
||||||
|
}
|
||||||
|
if !strings.Contains(result.stdout, "outputs=1 rejected=0") {
|
||||||
|
t.Fatalf("stdout=%q, want report counts", result.stdout)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunWritesTerminalArtifactsForResolutionPipelineAndOutputFailures(t *testing.T) {
|
||||||
|
for _, tc := range []struct {
|
||||||
|
name string
|
||||||
|
pipelineID string
|
||||||
|
wantError string
|
||||||
|
wantOutputs int
|
||||||
|
wantValidation string
|
||||||
|
configureFailure func(*testing.T, stateTestRoots, *stateTestHarness)
|
||||||
|
}{
|
||||||
|
{name: "resolution", pipelineID: "missing", wantError: `pipeline "missing"`},
|
||||||
|
{name: "pipeline", pipelineID: "sample", wantError: "synthetic extraction failure", wantValidation: "failed", configureFailure: func(_ *testing.T, _ stateTestRoots, h *stateTestHarness) {
|
||||||
|
h.extractErr = errors.New("synthetic extraction failure")
|
||||||
|
}},
|
||||||
|
{name: "output", pipelineID: "sample", wantError: "create output parent", wantOutputs: 1, wantValidation: "approved", configureFailure: func(t *testing.T, roots stateTestRoots, _ *stateTestHarness) {
|
||||||
|
if err := os.WriteFile(roots.output, []byte("not a directory"), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}},
|
||||||
|
} {
|
||||||
|
t.Run(tc.name, func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
if tc.configureFailure != nil {
|
||||||
|
tc.configureFailure(t, roots, harness)
|
||||||
|
}
|
||||||
|
opts := harness.options()
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
args := []string{"run", tc.pipelineID, "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass", "--debug"}
|
||||||
|
code := RunWithOptions(args, &stdout, &stderr, opts)
|
||||||
|
if code != 1 || !strings.Contains(stderr.String(), tc.wantError) {
|
||||||
|
t.Fatalf("code=%d stderr=%q", code, stderr.String())
|
||||||
|
}
|
||||||
|
bundlePath := onlyChildDir(t, roots.debug)
|
||||||
|
runID := filepath.Base(bundlePath)
|
||||||
|
report := readStateTestRunReport(t, bundlePath)
|
||||||
|
if report.Succeeded || report.RunID != runID || report.PipelineID != tc.pipelineID || report.OutputPath != filepath.Join(roots.output, runID) || report.DebugPath != bundlePath || report.OutputCount != tc.wantOutputs || report.RejectedCount != 0 || report.WarningCount != 0 || report.ValidationStatus != tc.wantValidation {
|
||||||
|
t.Fatalf("failure report = %#v", report)
|
||||||
|
}
|
||||||
|
errorLog, err := os.ReadFile(filepath.Join(bundlePath, "summary", "error.log"))
|
||||||
|
if err != nil || !strings.Contains(string(errorLog), tc.wantError) {
|
||||||
|
t.Fatalf("error log = %q, %v", errorLog, err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunRetainsPartialPipelineOutcomeInFailureSummary(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
harness.chunkWarnings = []contracts.Warning{{Scope: "chunk", ReasonCode: "partial-warning", Message: "warning retained before failure"}}
|
||||||
|
harness.extractErr = errors.New("synthetic partial pipeline failure")
|
||||||
|
|
||||||
|
result := runStateTest(t, roots, harness.options(), true, true, "bypass")
|
||||||
|
if result.code != 1 {
|
||||||
|
t.Fatalf("code=%d stderr=%q", result.code, result.stderr)
|
||||||
|
}
|
||||||
|
bundlePath := onlyChildDir(t, roots.debug)
|
||||||
|
report := readStateTestRunReport(t, bundlePath)
|
||||||
|
if report.Succeeded || report.OutputCount != 0 || report.RejectedCount != 0 || report.WarningCount != 1 || report.ValidationStatus != "failed" {
|
||||||
|
t.Fatalf("partial failure report = %#v", report)
|
||||||
|
}
|
||||||
|
|
||||||
|
var manifest artifacts.RunManifest
|
||||||
|
readStateTestSummaryJSON(t, bundlePath, "run-manifest.json", &manifest)
|
||||||
|
if manifest.RunID != report.RunID || manifest.PipelineID != "sample" || manifest.ValidationStatus != "failed" {
|
||||||
|
t.Fatalf("partial manifest = %#v", manifest)
|
||||||
|
}
|
||||||
|
var warnings []contracts.Warning
|
||||||
|
readStateTestSummaryJSON(t, bundlePath, "warnings.json", &warnings)
|
||||||
|
if len(warnings) != 1 || warnings[0].ReasonCode != "partial-warning" {
|
||||||
|
t.Fatalf("partial warnings = %#v", warnings)
|
||||||
|
}
|
||||||
|
var events []pipeline.CheckpointEvent
|
||||||
|
readStateTestSummaryJSON(t, bundlePath, "checkpoint-events.json", &events)
|
||||||
|
if len(events) == 0 || events[0].Stage != "source" {
|
||||||
|
t.Fatalf("partial checkpoint events = %#v, want retained source decision", events)
|
||||||
|
}
|
||||||
|
var chunkPlan artifacts.ChunkPlanSummary
|
||||||
|
readStateTestSummaryJSON(t, bundlePath, "chunk-plan.json", &chunkPlan)
|
||||||
|
if chunkPlan.Mode != "bypass" || chunkPlan.ValidationStatus == "not_run" {
|
||||||
|
t.Fatalf("partial chunk plan = %#v", chunkPlan)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunTerminalPersistenceFailuresDoNotRecurseOrHidePrimaryError(t *testing.T) {
|
||||||
|
for _, tc := range []struct {
|
||||||
|
name string
|
||||||
|
reportErr error
|
||||||
|
errorLogErr error
|
||||||
|
wantSecondary string
|
||||||
|
}{
|
||||||
|
{name: "run report", reportErr: errors.New("injected run report failure"), wantSecondary: "injected run report failure"},
|
||||||
|
{name: "error log", errorLogErr: errors.New("injected error log failure"), wantSecondary: "injected error log failure"},
|
||||||
|
} {
|
||||||
|
t.Run(tc.name, func(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
harness.extractErr = errors.New("primary pipeline failure")
|
||||||
|
opts := harness.options()
|
||||||
|
var terminal *recordingTerminalWriter
|
||||||
|
opts.DebugTerminalFactory = func(delegate *debugbundle.SummaryWriter) DebugTerminalWriter {
|
||||||
|
terminal = &recordingTerminalWriter{delegate: delegate, reportErr: tc.reportErr, errorLogErr: tc.errorLogErr}
|
||||||
|
return terminal
|
||||||
|
}
|
||||||
|
|
||||||
|
result := runStateTest(t, roots, opts, true, false, "bypass")
|
||||||
|
if result.code != 1 {
|
||||||
|
t.Fatalf("code=%d stderr=%q", result.code, result.stderr)
|
||||||
|
}
|
||||||
|
if terminal == nil {
|
||||||
|
t.Fatal("terminal writer was not constructed")
|
||||||
|
}
|
||||||
|
if terminal.reportCalls != 1 || terminal.errorLogCalls != 1 {
|
||||||
|
t.Fatalf("terminal calls = report:%d error:%d", terminal.reportCalls, terminal.errorLogCalls)
|
||||||
|
}
|
||||||
|
primaryIndex := strings.Index(result.stderr, "primary pipeline failure")
|
||||||
|
secondaryIndex := strings.Index(result.stderr, tc.wantSecondary)
|
||||||
|
debugIndex := strings.Index(result.stderr, "debug=")
|
||||||
|
if primaryIndex < 0 || secondaryIndex <= primaryIndex || debugIndex <= secondaryIndex {
|
||||||
|
t.Fatalf("stderr order = %q", result.stderr)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunReportFailureOnSuccessIsTerminalizedWithoutRetry(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
var terminal *recordingTerminalWriter
|
||||||
|
opts.DebugTerminalFactory = func(delegate *debugbundle.SummaryWriter) DebugTerminalWriter {
|
||||||
|
terminal = &recordingTerminalWriter{delegate: delegate, reportErr: errors.New("injected success report failure")}
|
||||||
|
return terminal
|
||||||
|
}
|
||||||
|
|
||||||
|
result := runStateTest(t, roots, opts, true, false, "bypass")
|
||||||
|
if result.code != 1 || !strings.Contains(result.stderr, "write debug run report") || !strings.Contains(result.stderr, "injected success report failure") {
|
||||||
|
t.Fatalf("code=%d stdout=%q stderr=%q", result.code, result.stdout, result.stderr)
|
||||||
|
}
|
||||||
|
if terminal == nil {
|
||||||
|
t.Fatal("terminal writer was not constructed")
|
||||||
|
}
|
||||||
|
if terminal.reportCalls != 1 || terminal.errorLogCalls != 1 {
|
||||||
|
t.Fatalf("terminal calls = report:%d error:%d", terminal.reportCalls, terminal.errorLogCalls)
|
||||||
|
}
|
||||||
|
if result.stdout != "" {
|
||||||
|
t.Fatalf("stdout=%q, want no success message", result.stdout)
|
||||||
|
}
|
||||||
|
bundlePath := onlyChildDir(t, roots.debug)
|
||||||
|
errorLog, err := os.ReadFile(filepath.Join(bundlePath, "summary", "error.log"))
|
||||||
|
if err != nil || !strings.Contains(string(errorLog), "injected success report failure") {
|
||||||
|
t.Fatalf("error log = %q, %v", errorLog, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunWithoutDebugDoesNotUseTerminalSummaryWriter(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
harness.extractErr = errors.New("non-debug pipeline failure")
|
||||||
|
opts := harness.options()
|
||||||
|
factoryCalls := 0
|
||||||
|
opts.DebugTerminalFactory = func(delegate *debugbundle.SummaryWriter) DebugTerminalWriter {
|
||||||
|
factoryCalls++
|
||||||
|
return delegate
|
||||||
|
}
|
||||||
|
|
||||||
|
result := runStateTest(t, roots, opts, false, false, "bypass")
|
||||||
|
if result.code != 1 || !strings.Contains(result.stderr, "non-debug pipeline failure") {
|
||||||
|
t.Fatalf("code=%d stderr=%q", result.code, result.stderr)
|
||||||
|
}
|
||||||
|
if factoryCalls != 0 {
|
||||||
|
t.Fatalf("terminal summary factory calls = %d, want 0", factoryCalls)
|
||||||
|
}
|
||||||
|
assertAbsent(t, roots.debug)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunRefusesExistingOutputDirectoryWithoutChangingIt(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
const runID = "run-1000000000-22222222222222222222222222222222"
|
||||||
|
runPath := filepath.Join(roots.output, runID)
|
||||||
|
if err := os.MkdirAll(filepath.Join(runPath, "nested"), 0o755); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(filepath.Join(runPath, "sentinel"), []byte("existing output"), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(filepath.Join(runPath, "nested", "data"), []byte("preserve me"), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
before := readTree(t, runPath)
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
opts.RunIDGenerator = func(time.Time) (string, error) { return runID, nil }
|
||||||
|
|
||||||
|
result := runStateTest(t, roots, opts, true, false, "bypass")
|
||||||
|
if result.code != 1 || !strings.Contains(result.stderr, "output run directory") || !strings.Contains(result.stderr, "already exists") {
|
||||||
|
t.Fatalf("code=%d stderr=%q", result.code, result.stderr)
|
||||||
|
}
|
||||||
|
if after := readTree(t, runPath); !sameFiles(after, before) {
|
||||||
|
t.Fatalf("existing output changed: before=%v after=%v", before, after)
|
||||||
|
}
|
||||||
|
bundlePath := filepath.Join(roots.debug, runID)
|
||||||
|
report := readStateTestRunReport(t, bundlePath)
|
||||||
|
if report.Succeeded || report.RunID != runID || report.OutputPath != runPath || report.DebugPath != bundlePath || report.OutputCount != 1 || report.ValidationStatus != "approved" {
|
||||||
|
t.Fatalf("output collision report = %#v", report)
|
||||||
|
}
|
||||||
|
errorLog, err := os.ReadFile(filepath.Join(bundlePath, "summary", "error.log"))
|
||||||
|
if err != nil || !strings.Contains(string(errorLog), "already exists") {
|
||||||
|
t.Fatalf("output collision error log = %q, %v", errorLog, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRepeatedRunIdentityCannotOverwriteFirstOutput(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
const runID = "run-1000000000-33333333333333333333333333333333"
|
||||||
|
harness := newStateTestHarness()
|
||||||
|
opts := harness.options()
|
||||||
|
opts.RunIDGenerator = func(time.Time) (string, error) { return runID, nil }
|
||||||
|
|
||||||
|
first := runStateTest(t, roots, opts, false, false, "bypass")
|
||||||
|
if first.code != 0 {
|
||||||
|
t.Fatalf("first code=%d stderr=%q", first.code, first.stderr)
|
||||||
|
}
|
||||||
|
runPath := filepath.Join(roots.output, runID)
|
||||||
|
before := readTree(t, runPath)
|
||||||
|
second := runStateTest(t, roots, opts, false, false, "bypass")
|
||||||
|
if second.code != 1 || !strings.Contains(second.stderr, "already exists") {
|
||||||
|
t.Fatalf("second code=%d stderr=%q", second.code, second.stderr)
|
||||||
|
}
|
||||||
|
if after := readTree(t, runPath); !sameFiles(after, before) {
|
||||||
|
t.Fatalf("first output changed: before=%v after=%v", before, after)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunRefusesExistingDebugBundleWithoutChangingIt(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
const runID = "run-1000000000-44444444444444444444444444444444"
|
||||||
|
bundlePath := filepath.Join(roots.debug, runID)
|
||||||
|
if err := os.MkdirAll(bundlePath, 0o700); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
sentinelPath := filepath.Join(bundlePath, "sentinel")
|
||||||
|
if err := os.WriteFile(sentinelPath, []byte("existing debug"), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
opts.RunIDGenerator = func(time.Time) (string, error) { return runID, nil }
|
||||||
|
|
||||||
|
result := runStateTest(t, roots, opts, true, false, "bypass")
|
||||||
|
if result.code != 1 || !strings.Contains(result.stderr, "debug bundle") || !strings.Contains(result.stderr, "already exists") {
|
||||||
|
t.Fatalf("code=%d stderr=%q", result.code, result.stderr)
|
||||||
|
}
|
||||||
|
if got, err := os.ReadFile(sentinelPath); err != nil || string(got) != "existing debug" {
|
||||||
|
t.Fatalf("sentinel = %q, %v", got, err)
|
||||||
|
}
|
||||||
|
assertAbsent(t, roots.output)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunIDGenerationFailurePrecedesDebugAllocation(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
opts.RunIDGenerator = func(time.Time) (string, error) { return "", errors.New("random source unavailable") }
|
||||||
|
|
||||||
|
result := runStateTest(t, roots, opts, true, false, "bypass")
|
||||||
|
if result.code != 1 || !strings.Contains(result.stderr, "generate run ID: random source unavailable") {
|
||||||
|
t.Fatalf("code=%d stderr=%q", result.code, result.stderr)
|
||||||
|
}
|
||||||
|
assertAbsent(t, roots.debug)
|
||||||
|
assertAbsent(t, roots.output)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunRejectsUnsafeGeneratedIdentityBeforePathUse(t *testing.T) {
|
||||||
|
roots := newStateTestRoots(t)
|
||||||
|
opts := newStateTestHarness().options()
|
||||||
|
opts.RunIDGenerator = func(time.Time) (string, error) { return "../outside", nil }
|
||||||
|
|
||||||
|
result := runStateTest(t, roots, opts, true, false, "bypass")
|
||||||
|
if result.code != 1 || !strings.Contains(result.stderr, "invalid generated run ID") || !strings.Contains(result.stderr, "one safe path component") {
|
||||||
|
t.Fatalf("code=%d stderr=%q", result.code, result.stderr)
|
||||||
|
}
|
||||||
|
assertAbsent(t, roots.debug)
|
||||||
|
assertAbsent(t, roots.output)
|
||||||
|
}
|
||||||
|
|
||||||
|
type stateTestRoots struct{ config, input, output, plans, checkpoints, debug string }
|
||||||
|
|
||||||
|
func newStateTestRoots(t *testing.T) stateTestRoots {
|
||||||
|
t.Helper()
|
||||||
|
base := t.TempDir()
|
||||||
|
roots := stateTestRoots{input: filepath.Join(base, "input.txt"), output: filepath.Join(base, "output"), plans: filepath.Join(base, "plans"), checkpoints: filepath.Join(base, "checkpoints"), debug: filepath.Join(base, "debug")}
|
||||||
|
if err := os.WriteFile(roots.input, []byte("application content Bearer secretvalue sk-secretvalue"), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
roots.config = filepath.Join(base, "config.yml")
|
||||||
|
config := fmt.Sprintf("version: 3\noutput:\n directory: %q\ncache:\n chunk_plans:\n directory: %q\n mode: auto\n checkpoints:\n enabled: true\n directory: %q\ndebug:\n directory: %q\npipelines:\n sample:\n input: test/input\n chunk: test/chunk\n artifacts:\n items:\n extract: test/extract\n merge: test/merge\n normalize: test/normalize\n output: test/output\n", roots.output, roots.plans, roots.checkpoints, roots.debug)
|
||||||
|
if err := os.WriteFile(roots.config, []byte(config), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return roots
|
||||||
|
}
|
||||||
|
|
||||||
|
type stateTestResult struct {
|
||||||
|
code int
|
||||||
|
stdout, stderr string
|
||||||
|
}
|
||||||
|
|
||||||
|
func runStateTest(t *testing.T, roots stateTestRoots, opts Options, debug, resume bool, mode string) stateTestResult {
|
||||||
|
t.Helper()
|
||||||
|
args := []string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", mode}
|
||||||
|
if debug {
|
||||||
|
args = append(args, "--debug")
|
||||||
|
}
|
||||||
|
if resume {
|
||||||
|
args = append(args, "--resume")
|
||||||
|
}
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
return stateTestResult{RunWithOptions(args, &stdout, &stderr, opts), stdout.String(), stderr.String()}
|
||||||
|
}
|
||||||
|
|
||||||
|
func assertStateTestOutput(t *testing.T, root string) {
|
||||||
|
t.Helper()
|
||||||
|
output := onlyChildDir(t, root)
|
||||||
|
data, err := os.ReadFile(filepath.Join(output, "result.json"))
|
||||||
|
if err != nil || string(data) != "{\"ok\":true}\n" {
|
||||||
|
t.Fatalf("output = %q, %v", data, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func onlyChildDir(t *testing.T, root string) string {
|
||||||
|
t.Helper()
|
||||||
|
entries, err := os.ReadDir(root)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var dirs []string
|
||||||
|
for _, entry := range entries {
|
||||||
|
if entry.IsDir() {
|
||||||
|
dirs = append(dirs, filepath.Join(root, entry.Name()))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(dirs) != 1 {
|
||||||
|
t.Fatalf("directories in %q = %v, want one", root, dirs)
|
||||||
|
}
|
||||||
|
return dirs[0]
|
||||||
|
}
|
||||||
|
|
||||||
|
func assertFile(t *testing.T, path string) {
|
||||||
|
t.Helper()
|
||||||
|
if info, err := os.Stat(path); err != nil || info.IsDir() {
|
||||||
|
t.Fatalf("file %q: %v", path, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
func assertAbsent(t *testing.T, path string) {
|
||||||
|
t.Helper()
|
||||||
|
if _, err := os.Stat(path); !os.IsNotExist(err) {
|
||||||
|
t.Fatalf("%q exists or stat failed: %v", path, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
func assertAnyFile(t *testing.T, root string) {
|
||||||
|
t.Helper()
|
||||||
|
if text := readAllFiles(t, root); text == "" {
|
||||||
|
t.Fatalf("no files under %q", root)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func readAllFiles(t *testing.T, root string) string {
|
||||||
|
t.Helper()
|
||||||
|
var content strings.Builder
|
||||||
|
if err := filepath.WalkDir(root, func(path string, entry os.DirEntry, err error) error {
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if entry.IsDir() {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
data, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
content.Write(data)
|
||||||
|
return nil
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return content.String()
|
||||||
|
}
|
||||||
|
|
||||||
|
func readStateTestRunReport(t *testing.T, bundlePath string) debugbundle.RunReport {
|
||||||
|
t.Helper()
|
||||||
|
var report debugbundle.RunReport
|
||||||
|
readStateTestSummaryJSON(t, bundlePath, "run-report.json", &report)
|
||||||
|
return report
|
||||||
|
}
|
||||||
|
|
||||||
|
func readStateTestSummaryJSON(t *testing.T, bundlePath, name string, target any) {
|
||||||
|
t.Helper()
|
||||||
|
data, err := os.ReadFile(filepath.Join(bundlePath, "summary", name))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := json.Unmarshal(data, target); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func assertRestrictedTree(t *testing.T, root string) {
|
||||||
|
t.Helper()
|
||||||
|
if runtime.GOOS == "windows" {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if err := filepath.Walk(root, func(path string, info os.FileInfo, err error) error {
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
want := os.FileMode(0o600)
|
||||||
|
if info.IsDir() {
|
||||||
|
want = 0o700
|
||||||
|
}
|
||||||
|
if info.Mode().Perm() != want {
|
||||||
|
return fmt.Errorf("%s has mode %o, want %o", path, info.Mode().Perm(), want)
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func readTree(t *testing.T, root string) map[string][]byte {
|
||||||
|
t.Helper()
|
||||||
|
files := map[string][]byte{}
|
||||||
|
if err := filepath.WalkDir(root, func(path string, entry os.DirEntry, err error) error {
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if entry.IsDir() {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
data, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
relative, err := filepath.Rel(root, path)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
files[relative] = data
|
||||||
|
return nil
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return files
|
||||||
|
}
|
||||||
|
func sameFiles(left, right map[string][]byte) bool {
|
||||||
|
if len(left) != len(right) {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
for path, data := range left {
|
||||||
|
if !bytes.Equal(data, right[path]) {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
|
type stateTestHarness struct {
|
||||||
|
mu sync.Mutex
|
||||||
|
chunkCalls, extractCalls int
|
||||||
|
runIDCalls uint64
|
||||||
|
extractErr error
|
||||||
|
chunkWarnings []contracts.Warning
|
||||||
|
moduleProfiles []string
|
||||||
|
sessionIDs []string
|
||||||
|
outputWarnings []contracts.Warning
|
||||||
|
includeWarnings bool
|
||||||
|
}
|
||||||
|
|
||||||
|
func newStateTestHarness() *stateTestHarness { return &stateTestHarness{} }
|
||||||
|
func (h *stateTestHarness) options() Options {
|
||||||
|
registries := pipeline.Registries{Inputs: pipeline.NewInputAdapterRegistry(), Chunkers: pipeline.NewChunkerRegistry(), ArtifactCodecs: pipeline.NewArtifactCodecRegistry(), Extractors: pipeline.NewExtractorRegistry(), Mergers: pipeline.NewMergerRegistry(), Normalizers: pipeline.NewNormalizerRegistry(), Validators: pipeline.NewValidatorRegistry(), ValidatorChains: pipeline.NewValidatorChainRegistry(), Outputs: pipeline.NewOutputEncoderRegistry()}
|
||||||
|
if err := pipeline.RegisterArtifactCodec(registries.ArtifactCodecs, stateTestCodec{}); err != nil {
|
||||||
|
panic(err)
|
||||||
|
}
|
||||||
|
if err := registries.Inputs.RegisterBuilderWithSpec(pipeline.ModuleSpec{Key: "test/input", Stage: pipeline.StageInput, Provides: []string{"source"}}, func(map[string]any) error { return nil }, func(pipeline.BuildRequest) (contracts.InputAdapter, error) { return stateTestInput{}, nil }); err != nil {
|
||||||
|
panic(err)
|
||||||
|
}
|
||||||
|
if err := registries.Chunkers.RegisterBuilderWithSpec(pipeline.ModuleSpec{Key: "test/chunk", Stage: pipeline.StageChunk, Requires: []string{"source"}, Provides: []string{"chunks"}, ReferenceSlots: []contracts.ReferenceSlot{{Name: "cache-reference"}}}, func(map[string]any) error { return nil }, func(pipeline.BuildRequest) (contracts.Chunker, error) { return stateTestChunker{h}, nil }); err != nil {
|
||||||
|
panic(err)
|
||||||
|
}
|
||||||
|
if err := pipeline.RegisterExtractor(registries.Extractors, pipeline.ModuleSpec{Key: "test/extract", Stage: pipeline.StageExtract, Requires: []string{"chunks"}, Provides: []string{"artifact"}, ArtifactKind: stateTestArtifactKind}, func() (contracts.Extractor[stateTestArtifact], error) { return stateTestExtractor{h}, nil }); err != nil {
|
||||||
|
panic(err)
|
||||||
|
}
|
||||||
|
if err := pipeline.RegisterMerger(registries.Mergers, pipeline.ModuleSpec{Key: "test/merge", Stage: pipeline.StageMerge, Requires: []string{"artifact"}, Provides: []string{"merged"}, ArtifactKind: stateTestArtifactKind}, func() (contracts.Merger[stateTestArtifact], error) { return stateTestMerger{harness: h}, nil }); err != nil {
|
||||||
|
panic(err)
|
||||||
|
}
|
||||||
|
if err := pipeline.RegisterNormalizer(registries.Normalizers, pipeline.ModuleSpec{Key: "test/normalize", Stage: pipeline.StageNormalize, Requires: []string{"merged"}, Provides: []string{"normalized"}, ArtifactKind: stateTestArtifactKind}, func() (contracts.Normalizer[stateTestArtifact], error) { return stateTestNormalizer{harness: h}, nil }); err != nil {
|
||||||
|
panic(err)
|
||||||
|
}
|
||||||
|
if err := registries.Outputs.RegisterWithSpec(pipeline.ModuleSpec{Key: "test/output", Stage: pipeline.StageOutput, Requires: []string{"normalized"}, Provides: []string{"output"}}, func() (contracts.OutputEncoder, error) {
|
||||||
|
return stateTestOutput{harness: h, includeWarnings: h.includeWarnings}, nil
|
||||||
|
}); err != nil {
|
||||||
|
panic(err)
|
||||||
|
}
|
||||||
|
return Options{Catalog: catalogFromRegistries(registries), Registries: registries, LookupEnv: emptyLookup, Now: func() time.Time { return time.Unix(1, 0) }, RunIDGenerator: func(startedAt time.Time) (string, error) {
|
||||||
|
h.mu.Lock()
|
||||||
|
defer h.mu.Unlock()
|
||||||
|
h.runIDCalls++
|
||||||
|
return fmt.Sprintf("run-%d-%032x", startedAt.UnixNano(), h.runIDCalls), nil
|
||||||
|
}, UserCacheDir: func() (string, error) { return "", errors.New("unexpected user cache lookup") }, LLMClientFactory: func(context.Context, config.Config, string) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
|
||||||
|
return nil, nil, nil
|
||||||
|
}}
|
||||||
|
}
|
||||||
|
|
||||||
|
type stateTestInput struct{}
|
||||||
|
|
||||||
|
func (stateTestInput) Key() string { return "test/input" }
|
||||||
|
func (stateTestInput) Parse(_ context.Context, req contracts.ParseRequest) (*source.SourceDocument, error) {
|
||||||
|
doc := &source.SourceDocument{ID: "source", Kind: "text", Format: "text/plain", Units: []source.SourceUnit{{ID: 1, Kind: "text", Text: string(req.Raw), Ref: source.SourceRef{SourceID: "source", StartUnitID: 1, EndUnitID: 1}}}}
|
||||||
|
digest, err := source.DigestDocument(doc)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
doc.Digest = digest
|
||||||
|
return doc, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type stateTestChunker struct{ harness *stateTestHarness }
|
||||||
|
|
||||||
|
func (stateTestChunker) Key() string { return "test/chunk" }
|
||||||
|
func (stateTestChunker) ReferenceSlots() []contracts.ReferenceSlot { return nil }
|
||||||
|
func (c stateTestChunker) Plan(_ context.Context, req contracts.ChunkRequest) (contracts.ChunkPlanResult, error) {
|
||||||
|
c.harness.mu.Lock()
|
||||||
|
c.harness.moduleProfiles = append(c.harness.moduleProfiles, req.LLMProfile)
|
||||||
|
c.harness.sessionIDs = append(c.harness.sessionIDs, req.SessionID)
|
||||||
|
c.harness.mu.Unlock()
|
||||||
|
c.harness.mu.Lock()
|
||||||
|
c.harness.chunkCalls++
|
||||||
|
c.harness.mu.Unlock()
|
||||||
|
return contracts.ChunkPlanResult{Plan: source.ChunkPlan{SourceDigest: req.Source.Digest, Ranges: []source.ChunkRange{{StartUnitID: 1, EndUnitID: 1}}}, Warnings: append([]contracts.Warning(nil), c.harness.chunkWarnings...)}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
const stateTestArtifactKind contracts.ArtifactKind = "test/artifact"
|
||||||
|
|
||||||
|
type stateTestArtifact struct {
|
||||||
|
Value string `json:"value"`
|
||||||
|
}
|
||||||
|
type stateTestCodec struct{}
|
||||||
|
|
||||||
|
func (stateTestCodec) Kind() contracts.ArtifactKind { return stateTestArtifactKind }
|
||||||
|
func (stateTestCodec) Schema() contracts.ArtifactSchema {
|
||||||
|
return contracts.ArtifactSchema{ID: "test.artifact", Name: "test_artifact", Version: "v1", JSONSchema: []byte(`{"type":"object"}`)}
|
||||||
|
}
|
||||||
|
func (stateTestCodec) MediaType() string { return "application/json" }
|
||||||
|
func (stateTestCodec) EncodeCandidate(v stateTestArtifact) ([]byte, error) {
|
||||||
|
return []byte(`{"value":"ok"}`), nil
|
||||||
|
}
|
||||||
|
func (stateTestCodec) Encode(v stateTestArtifact) ([]byte, error) {
|
||||||
|
return []byte(`{"value":"ok"}`), nil
|
||||||
|
}
|
||||||
|
func (stateTestCodec) Decode([]byte) (stateTestArtifact, error) {
|
||||||
|
return stateTestArtifact{Value: "ok"}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type stateTestExtractor struct{ harness *stateTestHarness }
|
||||||
|
|
||||||
|
func (stateTestExtractor) Key() string { return "test/extract" }
|
||||||
|
func (stateTestExtractor) ReferenceSlots() []contracts.ReferenceSlot { return nil }
|
||||||
|
func (e stateTestExtractor) Extract(_ context.Context, req contracts.TypedExtractionRequest) (contracts.TypedExtractionResult[stateTestArtifact], error) {
|
||||||
|
e.harness.mu.Lock()
|
||||||
|
defer e.harness.mu.Unlock()
|
||||||
|
e.harness.extractCalls++
|
||||||
|
e.harness.moduleProfiles = append(e.harness.moduleProfiles, req.LLMProfile)
|
||||||
|
e.harness.sessionIDs = append(e.harness.sessionIDs, req.SessionID)
|
||||||
|
if e.harness.extractErr != nil {
|
||||||
|
return contracts.TypedExtractionResult[stateTestArtifact]{}, e.harness.extractErr
|
||||||
|
}
|
||||||
|
return contracts.TypedExtractionResult[stateTestArtifact]{Value: stateTestArtifact{Value: "ok"}}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type stateTestMerger struct{ harness *stateTestHarness }
|
||||||
|
|
||||||
|
func (stateTestMerger) Key() string { return "test/merge" }
|
||||||
|
func (m stateTestMerger) Merge(_ context.Context, req contracts.TypedMergeRequest[stateTestArtifact]) (contracts.TypedMergeResult[stateTestArtifact], error) {
|
||||||
|
m.harness.mu.Lock()
|
||||||
|
m.harness.moduleProfiles = append(m.harness.moduleProfiles, req.LLMProfile)
|
||||||
|
m.harness.sessionIDs = append(m.harness.sessionIDs, req.SessionID)
|
||||||
|
m.harness.mu.Unlock()
|
||||||
|
return contracts.TypedMergeResult[stateTestArtifact]{Value: req.ExtractOutputs[0].Value}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type stateTestNormalizer struct{ harness *stateTestHarness }
|
||||||
|
|
||||||
|
func (stateTestNormalizer) Key() string { return "test/normalize" }
|
||||||
|
func (stateTestNormalizer) ReferenceSlots() []contracts.ReferenceSlot { return nil }
|
||||||
|
func (n stateTestNormalizer) Normalize(_ context.Context, req contracts.TypedNormalizeRequest[stateTestArtifact]) (contracts.TypedNormalizeResult[stateTestArtifact], error) {
|
||||||
|
n.harness.mu.Lock()
|
||||||
|
n.harness.moduleProfiles = append(n.harness.moduleProfiles, req.LLMProfile)
|
||||||
|
n.harness.sessionIDs = append(n.harness.sessionIDs, req.SessionID)
|
||||||
|
n.harness.mu.Unlock()
|
||||||
|
return contracts.TypedNormalizeResult[stateTestArtifact]{Value: req.MergeOutput.Value}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type stateTestOutput struct {
|
||||||
|
harness *stateTestHarness
|
||||||
|
includeWarnings bool
|
||||||
|
}
|
||||||
|
|
||||||
|
func (o stateTestOutput) Key() string { return "test/output" }
|
||||||
|
func (o stateTestOutput) Encode(_ context.Context, req contracts.OutputRequest) (contracts.OutputResult, error) {
|
||||||
|
o.harness.mu.Lock()
|
||||||
|
o.harness.outputWarnings = append([]contracts.Warning(nil), req.Warnings...)
|
||||||
|
o.harness.mu.Unlock()
|
||||||
|
data := []byte("{\"ok\":true}\n")
|
||||||
|
if o.includeWarnings && len(req.Warnings) > 0 {
|
||||||
|
data = []byte(fmt.Sprintf("{\"ok\":true,\"warnings\":%q}\n", req.Warnings[0].ReasonCode))
|
||||||
|
}
|
||||||
|
return contracts.OutputResult{Files: []contracts.OutputFile{{Name: "result.json", Bytes: data}}}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type failingDebugRecorder struct{}
|
||||||
|
|
||||||
|
func (failingDebugRecorder) Enabled() bool { return true }
|
||||||
|
func (failingDebugRecorder) WriteJSON(string, any) error { return errors.New("trace unavailable") }
|
||||||
|
func (failingDebugRecorder) WriteBytes(string, []byte) error { return errors.New("trace unavailable") }
|
||||||
|
|
||||||
|
type recordingTerminalWriter struct {
|
||||||
|
delegate DebugTerminalWriter
|
||||||
|
reportErr, errorLogErr error
|
||||||
|
reportCalls, errorLogCalls int
|
||||||
|
}
|
||||||
|
|
||||||
|
func (w *recordingTerminalWriter) WriteRunReport(report debugbundle.RunReport) error {
|
||||||
|
w.reportCalls++
|
||||||
|
if w.reportErr != nil {
|
||||||
|
return w.reportErr
|
||||||
|
}
|
||||||
|
return w.delegate.WriteRunReport(report)
|
||||||
|
}
|
||||||
|
|
||||||
|
func (w *recordingTerminalWriter) WriteError(message string) error {
|
||||||
|
w.errorLogCalls++
|
||||||
|
if w.errorLogErr != nil {
|
||||||
|
return w.errorLogErr
|
||||||
|
}
|
||||||
|
return w.delegate.WriteError(message)
|
||||||
|
}
|
||||||
@@ -1,91 +1,149 @@
|
|||||||
package artifacts
|
package artifacts
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"encoding/json"
|
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
type ArtifactCandidate struct {
|
|
||||||
Index int `json:"index"`
|
|
||||||
ExtractorKey string `json:"extractor_key"`
|
|
||||||
ArtifactType string `json:"artifact_type"`
|
|
||||||
SchemaVersion string `json:"schema_version"`
|
|
||||||
Payload json.RawMessage `json:"payload"`
|
|
||||||
SourceRefs []source.SourceRef `json:"source_refs,omitempty"`
|
|
||||||
Metadata map[string]any `json:"metadata,omitempty"`
|
|
||||||
}
|
|
||||||
|
|
||||||
type Artifact struct {
|
|
||||||
ExtractorKey string `json:"extractor_key"`
|
|
||||||
ArtifactType string `json:"artifact_type"`
|
|
||||||
SchemaVersion string `json:"schema_version"`
|
|
||||||
Payload json.RawMessage `json:"payload"`
|
|
||||||
SourceRefs []source.SourceRef `json:"source_refs,omitempty"`
|
|
||||||
Metadata map[string]any `json:"metadata,omitempty"`
|
|
||||||
}
|
|
||||||
|
|
||||||
type RejectedArtifact struct {
|
|
||||||
Candidate ArtifactCandidate `json:"candidate"`
|
|
||||||
ValidatorName string `json:"validator_name"`
|
|
||||||
ReasonCode string `json:"reason_code"`
|
|
||||||
Message string `json:"message"`
|
|
||||||
}
|
|
||||||
|
|
||||||
type ArtifactLaneManifest struct {
|
type ArtifactLaneManifest struct {
|
||||||
|
StepID string `json:"step_id,omitempty"`
|
||||||
ID string `json:"id"`
|
ID string `json:"id"`
|
||||||
Extractor string `json:"extractor"`
|
Extractor string `json:"extractor"`
|
||||||
Merger string `json:"merger"`
|
Merger string `json:"merger"`
|
||||||
Normalizer string `json:"normalizer"`
|
Normalizer string `json:"normalizer"`
|
||||||
Validators []string `json:"validators,omitempty"`
|
|
||||||
Metadata map[string]any `json:"metadata,omitempty"`
|
Metadata map[string]any `json:"metadata,omitempty"`
|
||||||
}
|
}
|
||||||
|
|
||||||
|
type ValidatorChainManifest struct {
|
||||||
|
Stage string `json:"stage"`
|
||||||
|
LaneID string `json:"lane_id,omitempty"`
|
||||||
|
ModuleKey string `json:"module_key"`
|
||||||
|
Validators []ValidatorManifest `json:"validators"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type ValidatorManifest struct {
|
||||||
|
Key string `json:"key"`
|
||||||
|
ExecutionClass string `json:"execution_class"`
|
||||||
|
}
|
||||||
|
|
||||||
type LLMProfileManifest struct {
|
type LLMProfileManifest struct {
|
||||||
ID string `json:"id"`
|
ID string `json:"id"`
|
||||||
Provider string `json:"provider,omitempty"`
|
Provider string `json:"provider,omitempty"`
|
||||||
Model string `json:"model,omitempty"`
|
Model string `json:"model,omitempty"`
|
||||||
}
|
}
|
||||||
|
|
||||||
|
type ReferenceProvenance struct {
|
||||||
|
Stage string `json:"stage,omitempty"`
|
||||||
|
StepID string `json:"step_id,omitempty"`
|
||||||
|
LaneID string `json:"lane_id,omitempty"`
|
||||||
|
SlotName string `json:"slot_name"`
|
||||||
|
OriginType string `json:"origin_type"`
|
||||||
|
OriginURI string `json:"origin_uri,omitempty"`
|
||||||
|
Digest string `json:"digest,omitempty"`
|
||||||
|
MediaType string `json:"media_type,omitempty"`
|
||||||
|
SizeBytes int64 `json:"size_bytes,omitempty"`
|
||||||
|
BindingSource string `json:"binding_source,omitempty"`
|
||||||
|
ArtifactKind string `json:"artifact_kind,omitempty"`
|
||||||
|
SchemaID string `json:"schema_id,omitempty"`
|
||||||
|
SchemaName string `json:"schema_name,omitempty"`
|
||||||
|
SchemaVersion string `json:"schema_version,omitempty"`
|
||||||
|
SchemaDigest string `json:"schema_digest,omitempty"`
|
||||||
|
ProducerPipeline string `json:"producer_pipeline_id,omitempty"`
|
||||||
|
ProducerStep string `json:"producer_step_id,omitempty"`
|
||||||
|
ProducerLane string `json:"producer_lane_id,omitempty"`
|
||||||
|
ProducerModule string `json:"producer_module_key,omitempty"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type OutputSchemaProvenance struct {
|
||||||
|
ID string `json:"id,omitempty"`
|
||||||
|
Name string `json:"name,omitempty"`
|
||||||
|
Version string `json:"version,omitempty"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type NormalizedOutputManifest struct {
|
||||||
|
StepID string `json:"step_id,omitempty"`
|
||||||
|
LaneID string `json:"lane_id"`
|
||||||
|
ModuleKey string `json:"module_key,omitempty"`
|
||||||
|
SourceID string `json:"source_id,omitempty"`
|
||||||
|
MediaType string `json:"media_type,omitempty"`
|
||||||
|
Schema OutputSchemaProvenance `json:"schema,omitempty"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type RejectedOutputManifest struct {
|
||||||
|
Stage string `json:"stage"`
|
||||||
|
StepID string `json:"step_id,omitempty"`
|
||||||
|
LaneID string `json:"lane_id,omitempty"`
|
||||||
|
ModuleKey string `json:"module_key,omitempty"`
|
||||||
|
ChunkID string `json:"chunk_id,omitempty"`
|
||||||
|
ChunkIndex int `json:"chunk_index,omitempty"`
|
||||||
|
ValidatorName string `json:"validator_name,omitempty"`
|
||||||
|
ReasonCode string `json:"reason_code,omitempty"`
|
||||||
|
Message string `json:"message,omitempty"`
|
||||||
|
AttemptCount int `json:"attempt_count,omitempty"`
|
||||||
|
DiagnosticArtifactPath string `json:"diagnostic_artifact_path,omitempty"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type CheckpointDecisionManifest struct {
|
||||||
|
Stage string `json:"stage"`
|
||||||
|
StepID string `json:"step_id,omitempty"`
|
||||||
|
LaneID string `json:"lane_id,omitempty"`
|
||||||
|
ModuleKey string `json:"module_key,omitempty"`
|
||||||
|
Category string `json:"category"`
|
||||||
|
ReasonCode string `json:"reason_code,omitempty"`
|
||||||
|
Detail string `json:"detail,omitempty"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type ChunkPlanManifest struct {
|
||||||
|
Mode string `json:"mode"`
|
||||||
|
Action string `json:"action,omitempty"`
|
||||||
|
SourceDigest string `json:"source_digest,omitempty"`
|
||||||
|
PlanDigest string `json:"plan_digest,omitempty"`
|
||||||
|
PlanSchemaVersion string `json:"plan_schema_version,omitempty"`
|
||||||
|
RequestedModule string `json:"requested_module"`
|
||||||
|
ProducerInputModule string `json:"producer_input_module,omitempty"`
|
||||||
|
ProducerModule string `json:"producer_module,omitempty"`
|
||||||
|
ProducerLLMProfile string `json:"producer_llm_profile,omitempty"`
|
||||||
|
ProducerReferences []ReferenceProvenance `json:"producer_references,omitempty"`
|
||||||
|
ProducerMetadata map[string]any `json:"producer_metadata,omitempty"`
|
||||||
|
CreatedAt *time.Time `json:"created_at,omitempty"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// ChunkPlanSummary is deliberately limited to cache and validation decisions.
|
||||||
|
// It must never contain plan units, source content, annotations, or model I/O.
|
||||||
|
type ChunkPlanSummary struct {
|
||||||
|
Mode string `json:"mode"`
|
||||||
|
SourceDigest string `json:"source_digest,omitempty"`
|
||||||
|
CandidateDigest string `json:"candidate_digest,omitempty"`
|
||||||
|
RequestedModule string `json:"requested_module"`
|
||||||
|
LookupStatus string `json:"lookup_status"`
|
||||||
|
LookupReason string `json:"lookup_reason,omitempty"`
|
||||||
|
Action string `json:"action,omitempty"`
|
||||||
|
ValidationStatus string `json:"validation_status"`
|
||||||
|
PublicationStatus string `json:"publication_status"`
|
||||||
|
}
|
||||||
|
|
||||||
type RunManifest struct {
|
type RunManifest struct {
|
||||||
RunID string `json:"run_id,omitempty"`
|
RunID string `json:"run_id,omitempty"`
|
||||||
PipelineID string `json:"pipeline_id,omitempty"`
|
PipelineID string `json:"pipeline_id,omitempty"`
|
||||||
PipelineDigest string `json:"pipeline_digest,omitempty"`
|
PipelineDigest string `json:"pipeline_digest,omitempty"`
|
||||||
InputModule string `json:"input_module,omitempty"`
|
InputModule string `json:"input_module,omitempty"`
|
||||||
Chunker string `json:"chunker,omitempty"`
|
Chunker string `json:"chunker,omitempty"`
|
||||||
|
ChunkPlan *ChunkPlanManifest `json:"chunk_plan,omitempty"`
|
||||||
SourceDigests []string `json:"source_digests,omitempty"`
|
SourceDigests []string `json:"source_digests,omitempty"`
|
||||||
Extractors []string `json:"extractors,omitempty"`
|
Extractors []string `json:"extractors,omitempty"`
|
||||||
Merger string `json:"merger,omitempty"`
|
Merger string `json:"merger,omitempty"`
|
||||||
Normalizer string `json:"normalizer,omitempty"`
|
Normalizer string `json:"normalizer,omitempty"`
|
||||||
OutputEncoder string `json:"output_encoder,omitempty"`
|
OutputEncoder string `json:"output_encoder,omitempty"`
|
||||||
|
ModuleMetadata map[string]map[string]any `json:"module_metadata,omitempty"`
|
||||||
ArtifactLanes []ArtifactLaneManifest `json:"artifact_lanes,omitempty"`
|
ArtifactLanes []ArtifactLaneManifest `json:"artifact_lanes,omitempty"`
|
||||||
|
ValidatorChains []ValidatorChainManifest `json:"validator_chains,omitempty"`
|
||||||
|
References []ReferenceProvenance `json:"references,omitempty"`
|
||||||
|
NormalizedOutputs []NormalizedOutputManifest `json:"normalized_outputs,omitempty"`
|
||||||
|
RejectedOutputs []RejectedOutputManifest `json:"rejected_outputs,omitempty"`
|
||||||
|
CheckpointDecisions []CheckpointDecisionManifest `json:"checkpoint_decisions,omitempty"`
|
||||||
LLMProfiles []LLMProfileManifest `json:"llm_profiles,omitempty"`
|
LLMProfiles []LLMProfileManifest `json:"llm_profiles,omitempty"`
|
||||||
|
Metadata map[string]any `json:"metadata,omitempty"`
|
||||||
SchemaVersion string `json:"schema_version,omitempty"`
|
SchemaVersion string `json:"schema_version,omitempty"`
|
||||||
ValidationStatus string `json:"validation_status,omitempty"`
|
ValidationStatus string `json:"validation_status,omitempty"`
|
||||||
StartedAt *time.Time `json:"started_at,omitempty"`
|
StartedAt *time.Time `json:"started_at,omitempty"`
|
||||||
CompletedAt *time.Time `json:"completed_at,omitempty"`
|
CompletedAt *time.Time `json:"completed_at,omitempty"`
|
||||||
}
|
}
|
||||||
|
|
||||||
func ArtifactFromCandidate(candidate ArtifactCandidate) Artifact {
|
|
||||||
return Artifact{
|
|
||||||
ExtractorKey: candidate.ExtractorKey,
|
|
||||||
ArtifactType: candidate.ArtifactType,
|
|
||||||
SchemaVersion: candidate.SchemaVersion,
|
|
||||||
Payload: append(json.RawMessage(nil), candidate.Payload...),
|
|
||||||
SourceRefs: append([]source.SourceRef(nil), candidate.SourceRefs...),
|
|
||||||
Metadata: copyMetadata(candidate.Metadata),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func copyMetadata(metadata map[string]any) map[string]any {
|
|
||||||
if len(metadata) == 0 {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
copied := make(map[string]any, len(metadata))
|
|
||||||
for key, value := range metadata {
|
|
||||||
copied[key] = value
|
|
||||||
}
|
|
||||||
return copied
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -2,116 +2,10 @@ package artifacts
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
"reflect"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
func TestArtifactFromCandidatePreservesCandidateFields(t *testing.T) {
|
|
||||||
candidate := ArtifactCandidate{
|
|
||||||
Index: 7,
|
|
||||||
ExtractorKey: "generic-extractor",
|
|
||||||
ArtifactType: "generic-artifact",
|
|
||||||
SchemaVersion: "v1",
|
|
||||||
Payload: json.RawMessage(`{"name":"example"}`),
|
|
||||||
SourceRefs: []source.SourceRef{
|
|
||||||
{SourceID: "source-1", StartUnitID: "u1", EndUnitID: "u2"},
|
|
||||||
},
|
|
||||||
Metadata: map[string]any{
|
|
||||||
"confidence": 0.75,
|
|
||||||
},
|
|
||||||
}
|
|
||||||
|
|
||||||
artifact := ArtifactFromCandidate(candidate)
|
|
||||||
|
|
||||||
if artifact.ExtractorKey != candidate.ExtractorKey {
|
|
||||||
t.Fatalf("ExtractorKey = %q, want %q", artifact.ExtractorKey, candidate.ExtractorKey)
|
|
||||||
}
|
|
||||||
if artifact.ArtifactType != candidate.ArtifactType {
|
|
||||||
t.Fatalf("ArtifactType = %q, want %q", artifact.ArtifactType, candidate.ArtifactType)
|
|
||||||
}
|
|
||||||
if artifact.SchemaVersion != candidate.SchemaVersion {
|
|
||||||
t.Fatalf("SchemaVersion = %q, want %q", artifact.SchemaVersion, candidate.SchemaVersion)
|
|
||||||
}
|
|
||||||
if string(artifact.Payload) != string(candidate.Payload) {
|
|
||||||
t.Fatalf("Payload = %s, want %s", artifact.Payload, candidate.Payload)
|
|
||||||
}
|
|
||||||
if !reflect.DeepEqual(artifact.SourceRefs, candidate.SourceRefs) {
|
|
||||||
t.Fatalf("SourceRefs = %#v, want %#v", artifact.SourceRefs, candidate.SourceRefs)
|
|
||||||
}
|
|
||||||
if !reflect.DeepEqual(artifact.Metadata, candidate.Metadata) {
|
|
||||||
t.Fatalf("Metadata = %#v, want %#v", artifact.Metadata, candidate.Metadata)
|
|
||||||
}
|
|
||||||
|
|
||||||
candidate.Payload[0] = '['
|
|
||||||
candidate.SourceRefs[0].StartUnitID = "changed"
|
|
||||||
candidate.Metadata["confidence"] = 0.5
|
|
||||||
|
|
||||||
if string(artifact.Payload) != `{"name":"example"}` {
|
|
||||||
t.Fatalf("Payload changed after candidate mutation: %s", artifact.Payload)
|
|
||||||
}
|
|
||||||
if artifact.SourceRefs[0].StartUnitID != "u1" {
|
|
||||||
t.Fatalf("SourceRefs changed after candidate mutation: %#v", artifact.SourceRefs)
|
|
||||||
}
|
|
||||||
if artifact.Metadata["confidence"] != 0.75 {
|
|
||||||
t.Fatalf("Metadata changed after candidate mutation: %#v", artifact.Metadata)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestJSONMarshalUsesExpectedFieldNames(t *testing.T) {
|
|
||||||
candidate := ArtifactCandidate{
|
|
||||||
Index: 1,
|
|
||||||
ExtractorKey: "generic-extractor",
|
|
||||||
ArtifactType: "generic-artifact",
|
|
||||||
SchemaVersion: "v1",
|
|
||||||
Payload: json.RawMessage(`{"value":true}`),
|
|
||||||
SourceRefs: []source.SourceRef{
|
|
||||||
{SourceID: "source-1", StartUnitID: "u1", EndUnitID: "u1"},
|
|
||||||
},
|
|
||||||
Metadata: map[string]any{
|
|
||||||
"reviewed": true,
|
|
||||||
},
|
|
||||||
}
|
|
||||||
rejected := RejectedArtifact{
|
|
||||||
Candidate: candidate,
|
|
||||||
ValidatorName: "generic-validator",
|
|
||||||
ReasonCode: "invalid",
|
|
||||||
Message: "candidate was not accepted",
|
|
||||||
}
|
|
||||||
|
|
||||||
gotJSON, err := json.Marshal(rejected)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("json.Marshal() error = %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
var got map[string]any
|
|
||||||
if err := json.Unmarshal(gotJSON, &got); err != nil {
|
|
||||||
t.Fatalf("json.Unmarshal() error = %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
assertHasKeys(t, got, "candidate", "validator_name", "reason_code", "message")
|
|
||||||
|
|
||||||
gotCandidate, ok := got["candidate"].(map[string]any)
|
|
||||||
if !ok {
|
|
||||||
t.Fatalf("candidate = %#v, want object", got["candidate"])
|
|
||||||
}
|
|
||||||
assertHasKeys(t, gotCandidate, "index", "extractor_key", "artifact_type", "schema_version", "payload", "source_refs", "metadata")
|
|
||||||
|
|
||||||
gotRefs, ok := gotCandidate["source_refs"].([]any)
|
|
||||||
if !ok {
|
|
||||||
t.Fatalf("source_refs = %#v, want array", gotCandidate["source_refs"])
|
|
||||||
}
|
|
||||||
if len(gotRefs) != 1 {
|
|
||||||
t.Fatalf("len(source_refs) = %d, want 1", len(gotRefs))
|
|
||||||
}
|
|
||||||
gotRef, ok := gotRefs[0].(map[string]any)
|
|
||||||
if !ok {
|
|
||||||
t.Fatalf("source_refs[0] = %#v, want object", gotRefs[0])
|
|
||||||
}
|
|
||||||
assertHasKeys(t, gotRef, "source_id", "start_unit_id", "end_unit_id")
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunManifestOmitsEmptyOptionalFields(t *testing.T) {
|
func TestRunManifestOmitsEmptyOptionalFields(t *testing.T) {
|
||||||
gotJSON, err := json.Marshal(RunManifest{})
|
gotJSON, err := json.Marshal(RunManifest{})
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -123,12 +17,43 @@ func TestRunManifestOmitsEmptyOptionalFields(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestRunManifestChunkPlanIsAdditiveAndOmitsPlanContent(t *testing.T) {
|
||||||
|
manifest := RunManifest{ChunkPlan: &ChunkPlanManifest{
|
||||||
|
Mode: "auto", Action: "reused", SourceDigest: "sha256:source", PlanDigest: "sha256:plan",
|
||||||
|
PlanSchemaVersion: "notarius.chunk-plan.v2", RequestedModule: "chunk/current",
|
||||||
|
ProducerInputModule: "input/original", ProducerModule: "chunk/original",
|
||||||
|
}}
|
||||||
|
encoded, err := json.Marshal(manifest)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
text := string(encoded)
|
||||||
|
for _, want := range []string{`"chunk_plan"`, `"action":"reused"`, `"requested_module":"chunk/current"`, `"producer_module":"chunk/original"`} {
|
||||||
|
if !strings.Contains(text, want) {
|
||||||
|
t.Fatalf("manifest JSON %s does not contain %s", text, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, forbidden := range []string{`"plan"`, `"units"`, `"annotations"`} {
|
||||||
|
if strings.Contains(text, forbidden) {
|
||||||
|
t.Fatalf("manifest JSON contains forbidden field %s: %s", forbidden, text)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
var legacy RunManifest
|
||||||
|
if err := json.Unmarshal([]byte(`{"pipeline_id":"legacy"}`), &legacy); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if legacy.PipelineID != "legacy" || legacy.ChunkPlan != nil {
|
||||||
|
t.Fatalf("legacy manifest = %#v", legacy)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestRunManifestIncludesPipelineAndArtifactLaneFields(t *testing.T) {
|
func TestRunManifestIncludesPipelineAndArtifactLaneFields(t *testing.T) {
|
||||||
manifest := RunManifest{
|
manifest := RunManifest{
|
||||||
PipelineID: "pipeline-1",
|
PipelineID: "pipeline-1",
|
||||||
PipelineDigest: "sha256:abc123",
|
PipelineDigest: "sha256:abc123",
|
||||||
LLMProfiles: []LLMProfileManifest{
|
LLMProfiles: []LLMProfileManifest{
|
||||||
{ID: "default", Provider: "openai-compatible", Model: "model-a"},
|
{ID: "default", Provider: "scriptorium", Model: "model-a"},
|
||||||
},
|
},
|
||||||
ArtifactLanes: []ArtifactLaneManifest{
|
ArtifactLanes: []ArtifactLaneManifest{
|
||||||
{
|
{
|
||||||
@@ -136,12 +61,21 @@ func TestRunManifestIncludesPipelineAndArtifactLaneFields(t *testing.T) {
|
|||||||
Extractor: "event-extractor",
|
Extractor: "event-extractor",
|
||||||
Merger: "appendorder",
|
Merger: "appendorder",
|
||||||
Normalizer: "noop",
|
Normalizer: "noop",
|
||||||
Validators: []string{"grounded"},
|
|
||||||
Metadata: map[string]any{
|
Metadata: map[string]any{
|
||||||
"extractor": map[string]any{"prompt_id": "test.prompt"},
|
"extractor": map[string]any{"prompt_id": "test.prompt"},
|
||||||
},
|
},
|
||||||
},
|
},
|
||||||
},
|
},
|
||||||
|
ValidatorChains: []ValidatorChainManifest{
|
||||||
|
{
|
||||||
|
Stage: "extract",
|
||||||
|
LaneID: "events",
|
||||||
|
ModuleKey: "event-extractor",
|
||||||
|
Validators: []ValidatorManifest{
|
||||||
|
{Key: "grounded", ExecutionClass: "deterministic"},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
gotJSON, err := json.Marshal(manifest)
|
gotJSON, err := json.Marshal(manifest)
|
||||||
@@ -154,7 +88,7 @@ func TestRunManifestIncludesPipelineAndArtifactLaneFields(t *testing.T) {
|
|||||||
t.Fatalf("json.Unmarshal() error = %v", err)
|
t.Fatalf("json.Unmarshal() error = %v", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
assertHasKeys(t, got, "pipeline_id", "pipeline_digest", "artifact_lanes", "llm_profiles")
|
assertHasKeys(t, got, "pipeline_id", "pipeline_digest", "artifact_lanes", "validator_chains", "llm_profiles")
|
||||||
|
|
||||||
profiles, ok := got["llm_profiles"].([]any)
|
profiles, ok := got["llm_profiles"].([]any)
|
||||||
if !ok {
|
if !ok {
|
||||||
@@ -180,7 +114,94 @@ func TestRunManifestIncludesPipelineAndArtifactLaneFields(t *testing.T) {
|
|||||||
if !ok {
|
if !ok {
|
||||||
t.Fatalf("artifact_lanes[0] = %#v, want object", lanes[0])
|
t.Fatalf("artifact_lanes[0] = %#v, want object", lanes[0])
|
||||||
}
|
}
|
||||||
assertHasKeys(t, lane, "id", "extractor", "merger", "normalizer", "validators", "metadata")
|
assertHasKeys(t, lane, "id", "extractor", "merger", "normalizer", "metadata")
|
||||||
|
|
||||||
|
chains, ok := got["validator_chains"].([]any)
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("validator_chains = %#v, want array", got["validator_chains"])
|
||||||
|
}
|
||||||
|
if len(chains) != 1 {
|
||||||
|
t.Fatalf("len(validator_chains) = %d, want 1", len(chains))
|
||||||
|
}
|
||||||
|
chain, ok := chains[0].(map[string]any)
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("validator_chains[0] = %#v, want object", chains[0])
|
||||||
|
}
|
||||||
|
assertHasKeys(t, chain, "stage", "lane_id", "module_key", "validators")
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunManifestIncludesReferenceProvenance(t *testing.T) {
|
||||||
|
manifest := RunManifest{
|
||||||
|
References: []ReferenceProvenance{
|
||||||
|
{
|
||||||
|
Stage: "extract",
|
||||||
|
LaneID: "events",
|
||||||
|
SlotName: "roster",
|
||||||
|
OriginType: "file",
|
||||||
|
OriginURI: "file:///tmp/roster.txt",
|
||||||
|
Digest: "sha256:reference",
|
||||||
|
MediaType: "text/plain; charset=utf-8",
|
||||||
|
SizeBytes: 12,
|
||||||
|
BindingSource: "config",
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
gotJSON, err := json.Marshal(manifest)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("json.Marshal() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
var got RunManifest
|
||||||
|
if err := json.Unmarshal(gotJSON, &got); err != nil {
|
||||||
|
t.Fatalf("json.Unmarshal() error = %v", err)
|
||||||
|
}
|
||||||
|
if len(got.References) != 1 {
|
||||||
|
t.Fatalf("len(References) = %d, want 1", len(got.References))
|
||||||
|
}
|
||||||
|
reference := got.References[0]
|
||||||
|
if reference.Stage != "extract" || reference.LaneID != "events" || reference.SlotName != "roster" || reference.OriginType != "file" || reference.OriginURI != "file:///tmp/roster.txt" {
|
||||||
|
t.Fatalf("reference provenance = %#v, want lane-scoped origin details", reference)
|
||||||
|
}
|
||||||
|
if reference.Digest != "sha256:reference" || reference.MediaType != "text/plain; charset=utf-8" || reference.SizeBytes != 12 || reference.BindingSource != "config" {
|
||||||
|
t.Fatalf("reference provenance = %#v, want digest/media/size/source details", reference)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunManifestIncludesTopLevelModuleMetadata(t *testing.T) {
|
||||||
|
manifest := RunManifest{
|
||||||
|
ModuleMetadata: map[string]map[string]any{
|
||||||
|
"chunker": {
|
||||||
|
"prompt_id": "dnd.scenes",
|
||||||
|
"prompt_version": "v1",
|
||||||
|
"prompt_sha256": "sha256:abc123",
|
||||||
|
"response_schema_key": "dnd_scenes",
|
||||||
|
"response_schema_name": "dnd_scenes",
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
gotJSON, err := json.Marshal(manifest)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("json.Marshal() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
var got map[string]any
|
||||||
|
if err := json.Unmarshal(gotJSON, &got); err != nil {
|
||||||
|
t.Fatalf("json.Unmarshal() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
moduleMetadata, ok := got["module_metadata"].(map[string]any)
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("module_metadata = %#v, want object", got["module_metadata"])
|
||||||
|
}
|
||||||
|
assertHasKeys(t, moduleMetadata, "chunker")
|
||||||
|
|
||||||
|
chunkerMetadata, ok := moduleMetadata["chunker"].(map[string]any)
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("module_metadata.chunker = %#v, want object", moduleMetadata["chunker"])
|
||||||
|
}
|
||||||
|
assertHasKeys(t, chunkerMetadata, "prompt_id", "prompt_version", "prompt_sha256", "response_schema_key", "response_schema_name")
|
||||||
}
|
}
|
||||||
|
|
||||||
func assertHasKeys(t *testing.T, values map[string]any, keys ...string) {
|
func assertHasKeys(t *testing.T, values map[string]any, keys ...string) {
|
||||||
|
|||||||
31
internal/core/config/cache.go
Normal file
31
internal/core/config/cache.go
Normal file
@@ -0,0 +1,31 @@
|
|||||||
|
package config
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
// DefaultChunkPlanRoot resolves the existing per-user chunk-plan cache root.
|
||||||
|
func DefaultChunkPlanRoot(userCacheDir func() (string, error)) (string, error) {
|
||||||
|
return defaultCacheFamilyRoot(userCacheDir, "chunk-plans")
|
||||||
|
}
|
||||||
|
|
||||||
|
func DefaultCheckpointRoot(userCacheDir func() (string, error)) (string, error) {
|
||||||
|
return defaultCacheFamilyRoot(userCacheDir, "checkpoints")
|
||||||
|
}
|
||||||
|
|
||||||
|
func defaultCacheFamilyRoot(userCacheDir func() (string, error), family string) (string, error) {
|
||||||
|
if userCacheDir == nil {
|
||||||
|
return "", fmt.Errorf("user cache directory resolver must not be nil")
|
||||||
|
}
|
||||||
|
root, err := userCacheDir()
|
||||||
|
if err != nil {
|
||||||
|
return "", fmt.Errorf("resolve user cache directory: %w", err)
|
||||||
|
}
|
||||||
|
root = strings.TrimSpace(root)
|
||||||
|
if root == "" {
|
||||||
|
return "", fmt.Errorf("user cache directory must not be empty")
|
||||||
|
}
|
||||||
|
return filepath.Join(filepath.Clean(root), "notarius", family), nil
|
||||||
|
}
|
||||||
@@ -1,66 +1,72 @@
|
|||||||
package config
|
package config
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/core/diagnostics"
|
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
)
|
)
|
||||||
|
|
||||||
const SupportedFileConfigVersion = 1
|
const SupportedFileConfigVersion = 3
|
||||||
|
|
||||||
type Config struct {
|
type Config struct {
|
||||||
LLMProfiles map[string]LLMProfile `json:"llm_profiles"`
|
Scriptorium ScriptoriumConfig `json:"scriptorium,omitempty"`
|
||||||
Pipelines map[string]pipeline.PipelineProfile `json:"pipelines"`
|
Pipelines map[string]pipeline.PipelineProfile `json:"pipelines"`
|
||||||
Concurrency ConcurrencyConfig `json:"concurrency"`
|
Concurrency ConcurrencyConfig `json:"concurrency"`
|
||||||
Diagnostics DiagnosticsConfig `json:"diagnostics"`
|
Output OutputConfig `json:"output"`
|
||||||
|
Cache CacheConfig `json:"cache"`
|
||||||
|
Debug DebugConfig `json:"debug"`
|
||||||
}
|
}
|
||||||
|
|
||||||
type LLMProfile struct {
|
type ScriptoriumConfig struct {
|
||||||
Provider string `json:"provider,omitempty"`
|
ProfileDir string `json:"profile_dir,omitempty"`
|
||||||
BaseURL string `json:"base_url,omitempty"`
|
ProfileFile string `json:"profile_file,omitempty"`
|
||||||
Model string `json:"model,omitempty"`
|
|
||||||
APIKey string `json:"api_key,omitempty"`
|
|
||||||
APIKeyEnv string `json:"api_key_env,omitempty"`
|
|
||||||
TimeoutSeconds int `json:"timeout_seconds,omitempty"`
|
|
||||||
MaxRetries int `json:"max_retries,omitempty"`
|
|
||||||
MaxConcurrency int `json:"max_concurrency,omitempty"`
|
|
||||||
}
|
}
|
||||||
|
|
||||||
type ConcurrencyConfig struct {
|
type ConcurrencyConfig struct {
|
||||||
TotalLLM int `json:"total_llm"`
|
TotalLLM int `json:"total_llm"`
|
||||||
|
StageWorkers map[string]int `json:"stage_workers"`
|
||||||
|
|
||||||
|
extractWorkersConfigured bool
|
||||||
|
defaultedExtractWorkers int
|
||||||
}
|
}
|
||||||
|
|
||||||
type DiagnosticsConfig struct {
|
type OutputConfig struct {
|
||||||
WorkDir string `json:"work_dir"`
|
Directory string `json:"directory"`
|
||||||
Retention diagnostics.RetentionMode `json:"retention"`
|
}
|
||||||
|
|
||||||
|
type CacheConfig struct {
|
||||||
|
ChunkPlans ChunkPlanCacheConfig `json:"chunk_plans"`
|
||||||
|
Checkpoints CheckpointCacheConfig `json:"checkpoints"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type ChunkPlanCacheConfig struct {
|
||||||
|
Directory string `json:"directory,omitempty"`
|
||||||
|
Mode pipeline.ChunkCacheMode `json:"mode"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type CheckpointCacheConfig struct {
|
||||||
|
Enabled bool `json:"enabled"`
|
||||||
|
Directory string `json:"directory,omitempty"`
|
||||||
|
}
|
||||||
|
type DebugConfig struct {
|
||||||
|
Directory string `json:"directory"`
|
||||||
}
|
}
|
||||||
|
|
||||||
func Default() Config {
|
func Default() Config {
|
||||||
return Config{
|
return Config{
|
||||||
LLMProfiles: map[string]LLMProfile{
|
|
||||||
pipeline.DefaultLLMProfile: {
|
|
||||||
Provider: "openai-compatible",
|
|
||||||
TimeoutSeconds: 600,
|
|
||||||
MaxRetries: 3,
|
|
||||||
MaxConcurrency: 1,
|
|
||||||
},
|
|
||||||
},
|
|
||||||
Pipelines: map[string]pipeline.PipelineProfile{},
|
Pipelines: map[string]pipeline.PipelineProfile{},
|
||||||
Concurrency: ConcurrencyConfig{
|
Concurrency: ConcurrencyConfig{
|
||||||
TotalLLM: 1,
|
TotalLLM: 1,
|
||||||
|
StageWorkers: map[string]int{"extract": 1},
|
||||||
|
defaultedExtractWorkers: 1,
|
||||||
},
|
},
|
||||||
Diagnostics: DiagnosticsConfig{
|
Output: OutputConfig{Directory: "./notarius-output"},
|
||||||
WorkDir: "/tmp/notarius",
|
Cache: CacheConfig{ChunkPlans: ChunkPlanCacheConfig{Mode: pipeline.ChunkCacheAuto}},
|
||||||
Retention: diagnostics.RetentionAuto,
|
Debug: DebugConfig{Directory: "./notarius-debug"},
|
||||||
},
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func cloneConfig(in Config) Config {
|
func cloneConfig(in Config) Config {
|
||||||
out := in
|
out := in
|
||||||
out.LLMProfiles = make(map[string]LLMProfile, len(in.LLMProfiles))
|
out.Concurrency.StageWorkers = cloneIntMap(in.Concurrency.StageWorkers)
|
||||||
for key, profile := range in.LLMProfiles {
|
|
||||||
out.LLMProfiles[key] = profile
|
|
||||||
}
|
|
||||||
out.Pipelines = make(map[string]pipeline.PipelineProfile, len(in.Pipelines))
|
out.Pipelines = make(map[string]pipeline.PipelineProfile, len(in.Pipelines))
|
||||||
for key, profile := range in.Pipelines {
|
for key, profile := range in.Pipelines {
|
||||||
out.Pipelines[key] = clonePipelineProfile(profile)
|
out.Pipelines[key] = clonePipelineProfile(profile)
|
||||||
@@ -68,11 +74,59 @@ func cloneConfig(in Config) Config {
|
|||||||
return out
|
return out
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func cloneIntMap(in map[string]int) map[string]int {
|
||||||
|
if len(in) == 0 {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
out := make(map[string]int, len(in))
|
||||||
|
for key, value := range in {
|
||||||
|
out[key] = value
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
func (c *ConcurrencyConfig) recomputeStageWorkerDefaults() {
|
||||||
|
if c == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if c.StageWorkers == nil {
|
||||||
|
c.StageWorkers = make(map[string]int)
|
||||||
|
}
|
||||||
|
if !c.extractWorkersConfigured {
|
||||||
|
if value, ok := c.StageWorkers["extract"]; ok && (c.defaultedExtractWorkers == 0 || value != c.defaultedExtractWorkers) {
|
||||||
|
c.extractWorkersConfigured = true
|
||||||
|
return
|
||||||
|
}
|
||||||
|
c.StageWorkers["extract"] = c.TotalLLM
|
||||||
|
c.defaultedExtractWorkers = c.TotalLLM
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func clonePipelineProfile(in pipeline.PipelineProfile) pipeline.PipelineProfile {
|
func clonePipelineProfile(in pipeline.PipelineProfile) pipeline.PipelineProfile {
|
||||||
out := in
|
out := in
|
||||||
out.Input = cloneModuleBinding(in.Input)
|
out.Input = cloneModuleBinding(in.Input)
|
||||||
out.Chunk = cloneModuleBinding(in.Chunk)
|
out.Chunk = cloneModuleBinding(in.Chunk)
|
||||||
out.Output = cloneModuleBinding(in.Output)
|
out.Output = cloneModuleBinding(in.Output)
|
||||||
|
out.References = cloneReferenceSourceMap(in.References)
|
||||||
|
if len(in.Artifacts) > 0 {
|
||||||
|
out.Artifacts = make(map[string]pipeline.ArtifactLaneProfile, len(in.Artifacts))
|
||||||
|
for key, lane := range in.Artifacts {
|
||||||
|
out.Artifacts[key] = cloneArtifactLaneProfile(lane)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if in.Steps != nil {
|
||||||
|
out.Steps = make([]pipeline.PipelineStepProfile, len(in.Steps))
|
||||||
|
for i, step := range in.Steps {
|
||||||
|
out.Steps[i] = clonePipelineStepProfile(step)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
func clonePipelineStepProfile(in pipeline.PipelineStepProfile) pipeline.PipelineStepProfile {
|
||||||
|
out := in
|
||||||
|
out.ID = in.ID
|
||||||
|
out.References = cloneReferenceSourceMap(in.References)
|
||||||
if len(in.Artifacts) > 0 {
|
if len(in.Artifacts) > 0 {
|
||||||
out.Artifacts = make(map[string]pipeline.ArtifactLaneProfile, len(in.Artifacts))
|
out.Artifacts = make(map[string]pipeline.ArtifactLaneProfile, len(in.Artifacts))
|
||||||
for key, lane := range in.Artifacts {
|
for key, lane := range in.Artifacts {
|
||||||
@@ -87,6 +141,7 @@ func cloneArtifactLaneProfile(in pipeline.ArtifactLaneProfile) pipeline.Artifact
|
|||||||
out.Extract = cloneModuleBinding(in.Extract)
|
out.Extract = cloneModuleBinding(in.Extract)
|
||||||
out.Merge = cloneModuleBinding(in.Merge)
|
out.Merge = cloneModuleBinding(in.Merge)
|
||||||
out.Normalize = cloneModuleBinding(in.Normalize)
|
out.Normalize = cloneModuleBinding(in.Normalize)
|
||||||
|
out.References = cloneReferenceSourceMap(in.References)
|
||||||
if len(in.Validators) > 0 {
|
if len(in.Validators) > 0 {
|
||||||
out.Validators = make([]pipeline.ModuleBinding, len(in.Validators))
|
out.Validators = make([]pipeline.ModuleBinding, len(in.Validators))
|
||||||
for i, binding := range in.Validators {
|
for i, binding := range in.Validators {
|
||||||
@@ -96,11 +151,55 @@ func cloneArtifactLaneProfile(in pipeline.ArtifactLaneProfile) pipeline.Artifact
|
|||||||
return out
|
return out
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func cloneStringMap(in map[string]string) map[string]string {
|
||||||
|
if len(in) == 0 {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
out := make(map[string]string, len(in))
|
||||||
|
for key, value := range in {
|
||||||
|
out[key] = value
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
func cloneReferenceSourceMap(in map[string]pipeline.ReferenceSource) map[string]pipeline.ReferenceSource {
|
||||||
|
if len(in) == 0 {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
out := make(map[string]pipeline.ReferenceSource, len(in))
|
||||||
|
for key, source := range in {
|
||||||
|
out[key] = cloneReferenceSource(source)
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
func cloneReferenceSource(in pipeline.ReferenceSource) pipeline.ReferenceSource {
|
||||||
|
out := in
|
||||||
|
if in.Artifact != nil {
|
||||||
|
artifact := *in.Artifact
|
||||||
|
out.Artifact = &artifact
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
func cloneModuleBinding(in pipeline.ModuleBinding) pipeline.ModuleBinding {
|
func cloneModuleBinding(in pipeline.ModuleBinding) pipeline.ModuleBinding {
|
||||||
out := in
|
out := in
|
||||||
if len(in.Options) > 0 {
|
if len(in.Options) > 0 {
|
||||||
out.Options = cloneOptions(in.Options)
|
out.Options = cloneOptions(in.Options)
|
||||||
}
|
}
|
||||||
|
out.References = cloneReferenceSourceMap(in.References)
|
||||||
|
out.Validators = cloneValidatorOverride(in.Validators)
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
func cloneValidatorOverride(in pipeline.ValidatorOverride) pipeline.ValidatorOverride {
|
||||||
|
out := pipeline.ValidatorOverride{Set: in.Set}
|
||||||
|
if len(in.Validators) > 0 {
|
||||||
|
out.Validators = make([]pipeline.ModuleBinding, len(in.Validators))
|
||||||
|
for i, binding := range in.Validators {
|
||||||
|
out.Validators[i] = cloneModuleBinding(binding)
|
||||||
|
}
|
||||||
|
}
|
||||||
return out
|
return out
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -1,78 +0,0 @@
|
|||||||
package config
|
|
||||||
|
|
||||||
import (
|
|
||||||
"testing"
|
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/core/diagnostics"
|
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
|
||||||
)
|
|
||||||
|
|
||||||
func TestDefaultValues(t *testing.T) {
|
|
||||||
cfg := Default()
|
|
||||||
|
|
||||||
defaultProfile, ok := cfg.LLMProfiles[pipeline.DefaultLLMProfile]
|
|
||||||
if !ok {
|
|
||||||
t.Fatalf("expected default LLM profile")
|
|
||||||
}
|
|
||||||
if defaultProfile.Provider != "openai-compatible" {
|
|
||||||
t.Fatalf("unexpected provider: %q", defaultProfile.Provider)
|
|
||||||
}
|
|
||||||
if defaultProfile.BaseURL != "" || defaultProfile.Model != "" {
|
|
||||||
t.Fatalf("default profile should not require base URL/model yet: %+v", defaultProfile)
|
|
||||||
}
|
|
||||||
if defaultProfile.TimeoutSeconds != 600 || defaultProfile.MaxRetries != 3 || defaultProfile.MaxConcurrency != 1 {
|
|
||||||
t.Fatalf("unexpected default LLM operational values: %+v", defaultProfile)
|
|
||||||
}
|
|
||||||
if len(cfg.Pipelines) != 0 {
|
|
||||||
t.Fatalf("expected no built-in pipeline profiles, got %v", cfg.Pipelines)
|
|
||||||
}
|
|
||||||
if cfg.Concurrency.TotalLLM != 1 {
|
|
||||||
t.Fatalf("unexpected total LLM concurrency: %d", cfg.Concurrency.TotalLLM)
|
|
||||||
}
|
|
||||||
if cfg.Diagnostics.WorkDir != "/tmp/notarius" {
|
|
||||||
t.Fatalf("unexpected diagnostics work dir: %q", cfg.Diagnostics.WorkDir)
|
|
||||||
}
|
|
||||||
if cfg.Diagnostics.Retention != diagnostics.RetentionAuto {
|
|
||||||
t.Fatalf("unexpected diagnostics retention: %q", cfg.Diagnostics.Retention)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestApplyFileConfigMergesWithDefaults(t *testing.T) {
|
|
||||||
fileCfg, err := ParseFileConfigYAML([]byte(`
|
|
||||||
version: 1
|
|
||||||
llm_profiles:
|
|
||||||
default:
|
|
||||||
model: test-model
|
|
||||||
pipelines:
|
|
||||||
example:
|
|
||||||
input: fake/input
|
|
||||||
artifacts:
|
|
||||||
events:
|
|
||||||
extract: fake/extract
|
|
||||||
`))
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("ParseFileConfigYAML: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
cfg := Default()
|
|
||||||
if err := cfg.applyFileConfigWithLookup(fileCfg, emptyLookup); err != nil {
|
|
||||||
t.Fatalf("ApplyFileConfig: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
profile := cfg.LLMProfiles[pipeline.DefaultLLMProfile]
|
|
||||||
if profile.Model != "test-model" {
|
|
||||||
t.Fatalf("expected file model, got %+v", profile)
|
|
||||||
}
|
|
||||||
if profile.Provider != "openai-compatible" || profile.TimeoutSeconds != 600 || profile.MaxRetries != 3 {
|
|
||||||
t.Fatalf("expected default LLM fields to be preserved, got %+v", profile)
|
|
||||||
}
|
|
||||||
if cfg.Concurrency.TotalLLM != 1 {
|
|
||||||
t.Fatalf("expected default concurrency preserved, got %d", cfg.Concurrency.TotalLLM)
|
|
||||||
}
|
|
||||||
if cfg.Diagnostics.Retention != diagnostics.RetentionAuto {
|
|
||||||
t.Fatalf("expected default diagnostics retention preserved, got %q", cfg.Diagnostics.Retention)
|
|
||||||
}
|
|
||||||
if _, ok := cfg.Pipelines["example"]; !ok {
|
|
||||||
t.Fatalf("expected file pipeline to be applied")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -3,9 +3,7 @@ package config
|
|||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
"strings"
|
"strings"
|
||||||
"time"
|
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -14,16 +12,21 @@ type ResolveInput struct {
|
|||||||
Only []string
|
Only []string
|
||||||
Catalog pipeline.ModuleCatalog
|
Catalog pipeline.ModuleCatalog
|
||||||
LLMProfileOverride string
|
LLMProfileOverride string
|
||||||
|
ReferenceOverrides []pipeline.ReferenceBinding
|
||||||
|
ReferenceUnbinds []pipeline.ReferenceUnbind
|
||||||
}
|
}
|
||||||
|
|
||||||
type EffectiveConfig struct {
|
type EffectiveConfig struct {
|
||||||
Config Config
|
Config Config
|
||||||
PipelineID string
|
PipelineID string
|
||||||
Only []string
|
Only []string
|
||||||
|
ReferenceOverrides []pipeline.ReferenceBinding
|
||||||
|
ReferenceUnbinds []pipeline.ReferenceUnbind
|
||||||
ResolvedPipeline pipeline.ResolvedPipeline
|
ResolvedPipeline pipeline.ResolvedPipeline
|
||||||
}
|
}
|
||||||
|
|
||||||
func (c Config) Resolve(input ResolveInput) (EffectiveConfig, error) {
|
func (c Config) Resolve(input ResolveInput) (EffectiveConfig, error) {
|
||||||
|
c.Concurrency.recomputeStageWorkerDefaults()
|
||||||
if err := c.Validate(); err != nil {
|
if err := c.Validate(); err != nil {
|
||||||
return EffectiveConfig{}, err
|
return EffectiveConfig{}, err
|
||||||
}
|
}
|
||||||
@@ -40,13 +43,14 @@ func (c Config) Resolve(input ResolveInput) (EffectiveConfig, error) {
|
|||||||
profile = clonePipelineProfile(profile)
|
profile = clonePipelineProfile(profile)
|
||||||
profile.ID = pipelineID
|
profile.ID = pipelineID
|
||||||
if override := strings.TrimSpace(input.LLMProfileOverride); override != "" {
|
if override := strings.TrimSpace(input.LLMProfileOverride); override != "" {
|
||||||
if !hasLLMProfile(c.LLMProfiles, override) {
|
|
||||||
return EffectiveConfig{}, fmt.Errorf("LLM profile override %q is not configured", override)
|
|
||||||
}
|
|
||||||
applyLLMProfileOverride(&profile, override)
|
applyLLMProfileOverride(&profile, override)
|
||||||
}
|
}
|
||||||
|
|
||||||
resolved, err := pipeline.ResolvePipeline(profile, pipeline.ResolveOptions{Only: input.Only}, input.Catalog)
|
resolved, err := pipeline.ResolvePipeline(profile, pipeline.ResolveOptions{
|
||||||
|
Only: input.Only,
|
||||||
|
ReferenceOverrides: append([]pipeline.ReferenceBinding(nil), input.ReferenceOverrides...),
|
||||||
|
ReferenceUnbinds: append([]pipeline.ReferenceUnbind(nil), input.ReferenceUnbinds...),
|
||||||
|
}, input.Catalog)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return EffectiveConfig{}, fmt.Errorf("resolve pipeline %q: %w", pipelineID, err)
|
return EffectiveConfig{}, fmt.Errorf("resolve pipeline %q: %w", pipelineID, err)
|
||||||
}
|
}
|
||||||
@@ -55,22 +59,25 @@ func (c Config) Resolve(input ResolveInput) (EffectiveConfig, error) {
|
|||||||
Config: cloneConfig(c),
|
Config: cloneConfig(c),
|
||||||
PipelineID: pipelineID,
|
PipelineID: pipelineID,
|
||||||
Only: append([]string(nil), input.Only...),
|
Only: append([]string(nil), input.Only...),
|
||||||
|
ReferenceOverrides: append([]pipeline.ReferenceBinding(nil), input.ReferenceOverrides...),
|
||||||
|
ReferenceUnbinds: append([]pipeline.ReferenceUnbind(nil), input.ReferenceUnbinds...),
|
||||||
ResolvedPipeline: resolved,
|
ResolvedPipeline: resolved,
|
||||||
}, nil
|
}, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func applyLLMProfileOverride(profile *pipeline.PipelineProfile, profileID string) {
|
func applyLLMProfileOverride(profile *pipeline.PipelineProfile, profileID string) {
|
||||||
profile.Input.LLMProfile = profileID
|
|
||||||
profile.Chunk.LLMProfile = profileID
|
profile.Chunk.LLMProfile = profileID
|
||||||
profile.Output.LLMProfile = profileID
|
apply := func(artifacts map[string]pipeline.ArtifactLaneProfile) {
|
||||||
for laneID, lane := range profile.Artifacts {
|
for laneID, lane := range artifacts {
|
||||||
lane.Extract.LLMProfile = profileID
|
lane.Extract.LLMProfile = profileID
|
||||||
lane.Merge.LLMProfile = profileID
|
lane.Merge.LLMProfile = profileID
|
||||||
lane.Normalize.LLMProfile = profileID
|
lane.Normalize.LLMProfile = profileID
|
||||||
for i := range lane.Validators {
|
artifacts[laneID] = lane
|
||||||
lane.Validators[i].LLMProfile = profileID
|
|
||||||
}
|
}
|
||||||
profile.Artifacts[laneID] = lane
|
}
|
||||||
|
apply(profile.Artifacts)
|
||||||
|
for index := range profile.Steps {
|
||||||
|
apply(profile.Steps[index].Artifacts)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -83,36 +90,3 @@ func lookupPipelineProfile(profiles map[string]pipeline.PipelineProfile, pipelin
|
|||||||
}
|
}
|
||||||
return pipeline.PipelineProfile{}, false
|
return pipeline.PipelineProfile{}, false
|
||||||
}
|
}
|
||||||
|
|
||||||
func (c Config) OpenAICompatibleClientConfig(profileID string) (llm.OpenAICompatibleClientConfig, error) {
|
|
||||||
trimmedID := strings.TrimSpace(profileID)
|
|
||||||
profile, ok := c.LLMProfile(trimmedID)
|
|
||||||
if !ok {
|
|
||||||
return llm.OpenAICompatibleClientConfig{}, fmt.Errorf("LLM profile %q is not configured", trimmedID)
|
|
||||||
}
|
|
||||||
|
|
||||||
provider := strings.TrimSpace(profile.Provider)
|
|
||||||
if provider == "" {
|
|
||||||
provider = providerOpenAICompatible
|
|
||||||
}
|
|
||||||
if provider != providerOpenAICompatible {
|
|
||||||
return llm.OpenAICompatibleClientConfig{}, fmt.Errorf("LLM profile %q provider %q is not supported", trimmedID, provider)
|
|
||||||
}
|
|
||||||
|
|
||||||
baseURL := strings.TrimSpace(profile.BaseURL)
|
|
||||||
if baseURL == "" {
|
|
||||||
return llm.OpenAICompatibleClientConfig{}, fmt.Errorf("LLM profile %q base URL must not be empty", trimmedID)
|
|
||||||
}
|
|
||||||
model := strings.TrimSpace(profile.Model)
|
|
||||||
if model == "" {
|
|
||||||
return llm.OpenAICompatibleClientConfig{}, fmt.Errorf("LLM profile %q model must not be empty", trimmedID)
|
|
||||||
}
|
|
||||||
|
|
||||||
return llm.OpenAICompatibleClientConfig{
|
|
||||||
BaseURL: baseURL,
|
|
||||||
Model: model,
|
|
||||||
APIKey: profile.APIKey,
|
|
||||||
MaxRetries: profile.MaxRetries,
|
|
||||||
RequestTimeout: time.Duration(profile.TimeoutSeconds) * time.Second,
|
|
||||||
}, nil
|
|
||||||
}
|
|
||||||
|
|||||||
514
internal/core/config/effective_config_contract_test.go
Normal file
514
internal/core/config/effective_config_contract_test.go
Normal file
@@ -0,0 +1,514 @@
|
|||||||
|
package config
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestEffectiveConfigRejectsEmptyAndUnknownPipelineIDs(t *testing.T) {
|
||||||
|
cfg := configForEffectiveTests(t, effectiveProfile())
|
||||||
|
for _, pipelineID := range []string{"", "missing"} {
|
||||||
|
name := pipelineID
|
||||||
|
if name == "" {
|
||||||
|
name = "empty"
|
||||||
|
}
|
||||||
|
t.Run(name, func(t *testing.T) {
|
||||||
|
_, err := cfg.Resolve(ResolveInput{PipelineID: pipelineID, Catalog: effectiveCatalog(t)})
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "pipeline") {
|
||||||
|
t.Fatalf("Resolve(%q) error = %v, want pipeline context", pipelineID, err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEffectiveConfigResolvesTrimmedPipelineMapKeys(t *testing.T) {
|
||||||
|
profile := effectiveProfile()
|
||||||
|
profile.ID = " main "
|
||||||
|
cfg := Default()
|
||||||
|
cfg.Pipelines = map[string]pipeline.PipelineProfile{" main ": profile}
|
||||||
|
effective, err := cfg.Resolve(ResolveInput{PipelineID: "main", Catalog: effectiveCatalog(t)})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Resolve() error = %v", err)
|
||||||
|
}
|
||||||
|
if effective.PipelineID != "main" || effective.ResolvedPipeline.ID != "main" {
|
||||||
|
t.Fatalf("resolved IDs = %q, %q", effective.PipelineID, effective.ResolvedPipeline.ID)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEffectiveConfigOnlySelectsRequestedLanesWithoutMutatingSource(t *testing.T) {
|
||||||
|
profile := effectiveProfile()
|
||||||
|
profile.Artifacts["other"] = pipeline.ArtifactLaneProfile{Extract: pipeline.Binding("extract")}
|
||||||
|
cfg := configForEffectiveTests(t, profile)
|
||||||
|
effective, err := cfg.Resolve(ResolveInput{
|
||||||
|
PipelineID: "main",
|
||||||
|
Only: []string{"other"},
|
||||||
|
Catalog: effectiveCatalog(t),
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Resolve() error = %v", err)
|
||||||
|
}
|
||||||
|
if len(effective.ResolvedPipeline.Steps[0].ArtifactLanes) != 1 || effective.ResolvedPipeline.Steps[0].ArtifactLanes[0].ID != "other" {
|
||||||
|
t.Fatalf("resolved lanes = %#v", effective.ResolvedPipeline.Steps[0].ArtifactLanes)
|
||||||
|
}
|
||||||
|
if len(cfg.Pipelines["main"].Artifacts) != 2 {
|
||||||
|
t.Fatalf("source lanes were mutated: %#v", cfg.Pipelines["main"].Artifacts)
|
||||||
|
}
|
||||||
|
|
||||||
|
_, err = cfg.Resolve(ResolveInput{
|
||||||
|
PipelineID: "main",
|
||||||
|
Only: []string{"missing"},
|
||||||
|
Catalog: effectiveCatalog(t),
|
||||||
|
})
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "lane \"missing\"") {
|
||||||
|
t.Fatalf("unknown lane error = %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEffectiveConfigMaterializesDefaultBindingsThroughCatalog(t *testing.T) {
|
||||||
|
effective, err := resolveEffectiveProfile(t, effectiveProfile(), ResolveInput{})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Resolve() error = %v", err)
|
||||||
|
}
|
||||||
|
resolved := effective.ResolvedPipeline
|
||||||
|
if resolved.Chunk.Module != pipeline.DefaultChunkModule || resolved.Output.Module != pipeline.DefaultOutputModule {
|
||||||
|
t.Fatalf("default pipeline bindings = %#v, %#v", resolved.Chunk, resolved.Output)
|
||||||
|
}
|
||||||
|
if len(resolved.Steps[0].ArtifactLanes) != 1 || resolved.Steps[0].ArtifactLanes[0].Merge.Module != pipeline.DefaultMergeModule || resolved.Steps[0].ArtifactLanes[0].Normalize.Module != pipeline.DefaultNormalizeModule {
|
||||||
|
t.Fatalf("default lane bindings = %#v", resolved.Steps[0].ArtifactLanes)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEffectiveConfigResolutionFailuresRetainContext(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
mutate func(*pipeline.PipelineProfile)
|
||||||
|
want []string
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
name: "unknown module",
|
||||||
|
mutate: func(profile *pipeline.PipelineProfile) {
|
||||||
|
profile.Input.Module = "missing-input"
|
||||||
|
},
|
||||||
|
want: []string{"pipeline \"main\"", "input"},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "missing capability",
|
||||||
|
mutate: func(profile *pipeline.PipelineProfile) {
|
||||||
|
profile.Chunk.Module = "needs-capability"
|
||||||
|
},
|
||||||
|
want: []string{"pipeline \"main\"", "chunk"},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "missing artifact variant",
|
||||||
|
mutate: func(profile *pipeline.PipelineProfile) {
|
||||||
|
profile.Artifacts["lane"] = pipeline.ArtifactLaneProfile{
|
||||||
|
Extract: pipeline.Binding("extract"),
|
||||||
|
Merge: pipeline.Binding("other-merge"),
|
||||||
|
}
|
||||||
|
},
|
||||||
|
want: []string{"pipeline \"main\"", "lane \"lane\"", "merge"},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "invalid module options",
|
||||||
|
mutate: func(profile *pipeline.PipelineProfile) {
|
||||||
|
profile.Chunk = pipeline.ModuleBinding{Module: "generic", Options: map[string]any{"unknown": true}}
|
||||||
|
},
|
||||||
|
want: []string{"pipeline \"main\"", "chunk", "generic", "options"},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "invalid validator options",
|
||||||
|
mutate: func(profile *pipeline.PipelineProfile) {
|
||||||
|
lane := profile.Artifacts["lane"]
|
||||||
|
lane.Extract.Validators = pipeline.ValidatorOverride{
|
||||||
|
Set: true,
|
||||||
|
Validators: []pipeline.ModuleBinding{{
|
||||||
|
Module: "option-validator",
|
||||||
|
Options: map[string]any{"invalid": true},
|
||||||
|
}},
|
||||||
|
}
|
||||||
|
profile.Artifacts["lane"] = lane
|
||||||
|
},
|
||||||
|
want: []string{"pipeline \"main\"", "lane \"lane\"", "extract", "option-validator", "options"},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
profile := effectiveProfile()
|
||||||
|
tt.mutate(&profile)
|
||||||
|
_, err := resolveEffectiveProfile(t, profile, ResolveInput{})
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("Resolve() error = nil, want failure")
|
||||||
|
}
|
||||||
|
for _, fragment := range tt.want {
|
||||||
|
if !strings.Contains(err.Error(), fragment) {
|
||||||
|
t.Fatalf("Resolve() error = %v, want context %q", err, fragment)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEffectiveConfigLLMProfileOverrideChangesDigestWithoutOverridingValidators(t *testing.T) {
|
||||||
|
profile := effectiveProfile()
|
||||||
|
profile.Chunk.LLMProfile = "chunk-profile"
|
||||||
|
lane := profile.Artifacts["lane"]
|
||||||
|
lane.Extract.LLMProfile = "extract-profile"
|
||||||
|
lane.Merge.LLMProfile = "merge-profile"
|
||||||
|
lane.Normalize.LLMProfile = "normalize-profile"
|
||||||
|
lane.Extract.Validators = pipeline.ValidatorOverride{
|
||||||
|
Set: true,
|
||||||
|
Validators: []pipeline.ModuleBinding{{
|
||||||
|
Module: "llm-validator",
|
||||||
|
LLMProfile: "validator-profile",
|
||||||
|
}},
|
||||||
|
}
|
||||||
|
profile.Artifacts["lane"] = lane
|
||||||
|
|
||||||
|
base, err := resolveEffectiveProfile(t, profile, ResolveInput{})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("base Resolve() error = %v", err)
|
||||||
|
}
|
||||||
|
overridden, err := resolveEffectiveProfile(t, profile, ResolveInput{LLMProfileOverride: "override-profile"})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("overridden Resolve() error = %v", err)
|
||||||
|
}
|
||||||
|
if base.ResolvedPipeline.Digest == overridden.ResolvedPipeline.Digest {
|
||||||
|
t.Fatal("LLM profile override did not change the pipeline digest")
|
||||||
|
}
|
||||||
|
resolved := overridden.ResolvedPipeline
|
||||||
|
if resolved.Chunk.LLMProfile != "override-profile" || resolved.Steps[0].ArtifactLanes[0].Extract.LLMProfile != "override-profile" ||
|
||||||
|
resolved.Steps[0].ArtifactLanes[0].Merge.LLMProfile != "override-profile" || resolved.Steps[0].ArtifactLanes[0].Normalize.LLMProfile != "override-profile" {
|
||||||
|
t.Fatalf("pipeline profile override was not applied: %#v", resolved)
|
||||||
|
}
|
||||||
|
validators := findEffectiveValidatorChain(resolved, pipeline.StageExtract, "lane")
|
||||||
|
if len(validators.Validators) != 1 || validators.Validators[0].Binding.LLMProfile != "validator-profile" {
|
||||||
|
t.Fatalf("validator profile was overridden: %#v", validators)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEffectiveConfigValidatorOverridesRemainDistinctAndOrdered(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
value pipeline.ValidatorOverride
|
||||||
|
want []string
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
name: "omitted uses default",
|
||||||
|
want: []string{"default-validator"},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "explicit empty",
|
||||||
|
value: pipeline.ValidatorOverride{Set: true},
|
||||||
|
want: nil,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "configured order",
|
||||||
|
value: pipeline.ValidatorOverride{
|
||||||
|
Set: true,
|
||||||
|
Validators: []pipeline.ModuleBinding{
|
||||||
|
pipeline.Binding("configured-a"),
|
||||||
|
pipeline.Binding("configured-b"),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
want: []string{"configured-a", "configured-b"},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
profile := effectiveProfile()
|
||||||
|
lane := profile.Artifacts["lane"]
|
||||||
|
lane.Extract.Validators = tt.value
|
||||||
|
profile.Artifacts["lane"] = lane
|
||||||
|
effective, err := resolveEffectiveProfile(t, profile, ResolveInput{})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Resolve() error = %v", err)
|
||||||
|
}
|
||||||
|
chain := findEffectiveValidatorChain(effective.ResolvedPipeline, pipeline.StageExtract, "lane")
|
||||||
|
got := make([]string, len(chain.Validators))
|
||||||
|
for i, validator := range chain.Validators {
|
||||||
|
got[i] = validator.Binding.Module
|
||||||
|
}
|
||||||
|
if len(got) != len(tt.want) {
|
||||||
|
t.Fatalf("validator chain = %#v, want %v", got, tt.want)
|
||||||
|
}
|
||||||
|
for i := range got {
|
||||||
|
if got[i] != tt.want[i] {
|
||||||
|
t.Fatalf("validator chain = %#v, want %v", got, tt.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEffectiveConfigAndResolutionInputsDoNotAliasSource(t *testing.T) {
|
||||||
|
profile := effectiveProfile()
|
||||||
|
profile.Chunk.Options = map[string]any{"nested": map[string]any{"safe": "source"}}
|
||||||
|
profile.Chunk.References = pipeline.ExternalReferenceMap(map[string]string{"chunk-ref": "chunk.txt"})
|
||||||
|
lane := profile.Artifacts["lane"]
|
||||||
|
lane.Extract.Validators = pipeline.ValidatorOverride{
|
||||||
|
Set: true,
|
||||||
|
Validators: []pipeline.ModuleBinding{{
|
||||||
|
Module: "configured-a",
|
||||||
|
Options: map[string]any{"nested": map[string]any{"safe": "validator-source"}},
|
||||||
|
}},
|
||||||
|
}
|
||||||
|
profile.Artifacts["lane"] = lane
|
||||||
|
cfg := configForEffectiveTests(t, profile)
|
||||||
|
only := []string{"lane"}
|
||||||
|
overrides := []pipeline.ReferenceBinding{{Stage: pipeline.StageChunk, SlotName: "chunk-ref", Source: "source.txt"}}
|
||||||
|
effective, err := cfg.Resolve(ResolveInput{
|
||||||
|
PipelineID: "main",
|
||||||
|
Only: only,
|
||||||
|
ReferenceOverrides: overrides,
|
||||||
|
Catalog: effectiveCatalog(t),
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Resolve() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
effective.Config.Pipelines["main"].Chunk.Options["nested"].(map[string]any)["safe"] = "effective-config"
|
||||||
|
effective.ResolvedPipeline.Chunk.Options["nested"].(map[string]any)["safe"] = "resolved-pipeline"
|
||||||
|
effective.ResolvedPipeline.ChunkReferences.Bindings[0].Source = "resolved-reference"
|
||||||
|
effective.ResolvedPipeline.ValidatorChains[1].Validators[0].Binding.Options["nested"].(map[string]any)["safe"] = "resolved-validator"
|
||||||
|
effective.Only[0] = "mutated-only"
|
||||||
|
effective.ReferenceOverrides[0].Source = "mutated-override"
|
||||||
|
|
||||||
|
if got := cfg.Pipelines["main"].Chunk.Options["nested"].(map[string]any)["safe"]; got != "source" {
|
||||||
|
t.Fatalf("source config option was aliased: %v", got)
|
||||||
|
}
|
||||||
|
if got := cfg.Pipelines["main"].Chunk.References["chunk-ref"].Path; got != "chunk.txt" {
|
||||||
|
t.Fatalf("source config references were aliased: %v", got)
|
||||||
|
}
|
||||||
|
if only[0] != "lane" || overrides[0].Source != "source.txt" {
|
||||||
|
t.Fatal("resolution inputs were aliased")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func effectiveProfile() pipeline.PipelineProfile {
|
||||||
|
return pipeline.PipelineProfile{
|
||||||
|
ID: "main",
|
||||||
|
Input: pipeline.Binding("input"),
|
||||||
|
Artifacts: map[string]pipeline.ArtifactLaneProfile{
|
||||||
|
"lane": {Extract: pipeline.Binding("extract")},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func configForEffectiveTests(t *testing.T, profile pipeline.PipelineProfile) Config {
|
||||||
|
t.Helper()
|
||||||
|
cfg := Default()
|
||||||
|
cfg.Pipelines = map[string]pipeline.PipelineProfile{"main": profile}
|
||||||
|
return cfg
|
||||||
|
}
|
||||||
|
|
||||||
|
func resolveEffectiveProfile(t *testing.T, profile pipeline.PipelineProfile, input ResolveInput) (EffectiveConfig, error) {
|
||||||
|
t.Helper()
|
||||||
|
cfg := configForEffectiveTests(t, profile)
|
||||||
|
if input.PipelineID == "" {
|
||||||
|
input.PipelineID = "main"
|
||||||
|
}
|
||||||
|
if input.Catalog.Inputs == nil {
|
||||||
|
input.Catalog = effectiveCatalog(t)
|
||||||
|
}
|
||||||
|
return cfg.Resolve(input)
|
||||||
|
}
|
||||||
|
|
||||||
|
func findEffectiveValidatorChain(resolved pipeline.ResolvedPipeline, stage pipeline.ModuleStage, laneID string) pipeline.ResolvedValidatorChain {
|
||||||
|
for _, chain := range resolved.ValidatorChains {
|
||||||
|
if chain.Stage == stage && chain.LaneID == laneID {
|
||||||
|
return chain
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return pipeline.ResolvedValidatorChain{}
|
||||||
|
}
|
||||||
|
|
||||||
|
type effectiveArtifact struct {
|
||||||
|
Value string `json:"value"`
|
||||||
|
}
|
||||||
|
|
||||||
|
const effectiveArtifactKind contracts.ArtifactKind = "test/effective"
|
||||||
|
|
||||||
|
type effectiveCodec struct{}
|
||||||
|
|
||||||
|
func (effectiveCodec) Kind() contracts.ArtifactKind { return effectiveArtifactKind }
|
||||||
|
func (effectiveCodec) Schema() contracts.ArtifactSchema {
|
||||||
|
return contracts.ArtifactSchema{
|
||||||
|
ID: "effective-schema",
|
||||||
|
Name: "Effective artifact",
|
||||||
|
Version: "1",
|
||||||
|
JSONSchema: []byte(`{"type":"object"}`),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
func (effectiveCodec) MediaType() string { return "application/json" }
|
||||||
|
func (effectiveCodec) EncodeCandidate(value effectiveArtifact) ([]byte, error) {
|
||||||
|
return json.Marshal(value)
|
||||||
|
}
|
||||||
|
func (effectiveCodec) Encode(value effectiveArtifact) ([]byte, error) {
|
||||||
|
return json.Marshal(value)
|
||||||
|
}
|
||||||
|
func (effectiveCodec) Decode(content []byte) (effectiveArtifact, error) {
|
||||||
|
var value effectiveArtifact
|
||||||
|
err := json.Unmarshal(content, &value)
|
||||||
|
return value, err
|
||||||
|
}
|
||||||
|
|
||||||
|
type effectiveInput struct{ key string }
|
||||||
|
|
||||||
|
func (m effectiveInput) Key() string { return m.key }
|
||||||
|
func (m effectiveInput) Parse(context.Context, contracts.ParseRequest) (*source.SourceDocument, error) {
|
||||||
|
return &source.SourceDocument{}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type effectiveChunker struct{ key string }
|
||||||
|
|
||||||
|
func (m effectiveChunker) Key() string { return m.key }
|
||||||
|
func (m effectiveChunker) ReferenceSlots() []contracts.ReferenceSlot {
|
||||||
|
if m.key == pipeline.DefaultChunkModule {
|
||||||
|
return []contracts.ReferenceSlot{{Name: "chunk-ref"}}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
func (m effectiveChunker) Plan(context.Context, contracts.ChunkRequest) (contracts.ChunkPlanResult, error) {
|
||||||
|
return contracts.ChunkPlanResult{}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type effectiveExtractor struct{ key string }
|
||||||
|
|
||||||
|
func (m effectiveExtractor) Key() string { return m.key }
|
||||||
|
func (m effectiveExtractor) ReferenceSlots() []contracts.ReferenceSlot { return nil }
|
||||||
|
func (m effectiveExtractor) Extract(context.Context, contracts.TypedExtractionRequest) (contracts.TypedExtractionResult[effectiveArtifact], error) {
|
||||||
|
return contracts.TypedExtractionResult[effectiveArtifact]{}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type effectiveMerger struct{ key string }
|
||||||
|
|
||||||
|
func (m effectiveMerger) Key() string { return m.key }
|
||||||
|
func (m effectiveMerger) Merge(context.Context, contracts.TypedMergeRequest[effectiveArtifact]) (contracts.TypedMergeResult[effectiveArtifact], error) {
|
||||||
|
return contracts.TypedMergeResult[effectiveArtifact]{}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type effectiveNormalizer struct{ key string }
|
||||||
|
|
||||||
|
func (m effectiveNormalizer) Key() string { return m.key }
|
||||||
|
func (m effectiveNormalizer) ReferenceSlots() []contracts.ReferenceSlot { return nil }
|
||||||
|
func (m effectiveNormalizer) Normalize(context.Context, contracts.TypedNormalizeRequest[effectiveArtifact]) (contracts.TypedNormalizeResult[effectiveArtifact], error) {
|
||||||
|
return contracts.TypedNormalizeResult[effectiveArtifact]{}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type effectiveOutput struct{ key string }
|
||||||
|
|
||||||
|
func (m effectiveOutput) Key() string { return m.key }
|
||||||
|
func (m effectiveOutput) Encode(context.Context, contracts.OutputRequest) (contracts.OutputResult, error) {
|
||||||
|
return contracts.OutputResult{}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type effectiveValidator struct {
|
||||||
|
name string
|
||||||
|
class contracts.ExecutionClass
|
||||||
|
}
|
||||||
|
|
||||||
|
func (v effectiveValidator) Name() string { return v.name }
|
||||||
|
func (v effectiveValidator) ExecutionClass() contracts.ExecutionClass { return v.class }
|
||||||
|
func (v effectiveValidator) Validate(context.Context, contracts.TypedValidationRequest[effectiveArtifact]) (contracts.ValidationResult, error) {
|
||||||
|
return contracts.ValidationResult{Approved: true}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func effectiveCatalog(t *testing.T) pipeline.ModuleCatalog {
|
||||||
|
t.Helper()
|
||||||
|
catalog := pipeline.ModuleCatalog{
|
||||||
|
Inputs: pipeline.NewInputAdapterRegistry(),
|
||||||
|
Chunkers: pipeline.NewChunkerRegistry(),
|
||||||
|
ArtifactCodecs: pipeline.NewArtifactCodecRegistry(),
|
||||||
|
Extractors: pipeline.NewExtractorRegistry(),
|
||||||
|
Mergers: pipeline.NewMergerRegistry(),
|
||||||
|
Normalizers: pipeline.NewNormalizerRegistry(),
|
||||||
|
Validators: pipeline.NewValidatorRegistry(),
|
||||||
|
ValidatorChains: pipeline.NewValidatorChainRegistry(),
|
||||||
|
Outputs: pipeline.NewOutputEncoderRegistry(),
|
||||||
|
}
|
||||||
|
if err := pipeline.RegisterArtifactCodec(catalog.ArtifactCodecs, effectiveCodec{}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := catalog.Inputs.RegisterWithSpec(pipeline.ModuleSpec{Key: "input", Stage: pipeline.StageInput, Provides: []string{"source"}}, func() (contracts.InputAdapter, error) {
|
||||||
|
return effectiveInput{key: "input"}, nil
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
chunkSpec := pipeline.ModuleSpec{
|
||||||
|
Key: pipeline.DefaultChunkModule,
|
||||||
|
Stage: pipeline.StageChunk,
|
||||||
|
Requires: []string{"source"},
|
||||||
|
Provides: []string{"chunk"},
|
||||||
|
ReferenceSlots: []contracts.ReferenceSlot{{Name: "chunk-ref"}},
|
||||||
|
}
|
||||||
|
chunkOptions := func(options map[string]any) error { return pipeline.RejectUnknownOptions(options, "size", "nested") }
|
||||||
|
if err := catalog.Chunkers.RegisterBuilderWithSpec(chunkSpec, chunkOptions, func(pipeline.BuildRequest) (contracts.Chunker, error) {
|
||||||
|
return effectiveChunker{key: pipeline.DefaultChunkModule}, nil
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := catalog.Chunkers.RegisterBuilderWithSpec(pipeline.ModuleSpec{Key: "needs-capability", Stage: pipeline.StageChunk, Requires: []string{"missing"}}, chunkOptions, func(pipeline.BuildRequest) (contracts.Chunker, error) {
|
||||||
|
return effectiveChunker{key: "needs-capability"}, nil
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := pipeline.RegisterExtractor(catalog.Extractors, pipeline.ModuleSpec{Key: "extract", Stage: pipeline.StageExtract, ArtifactKind: effectiveArtifactKind, Requires: []string{"chunk"}, Provides: []string{"candidate"}}, func() (contracts.Extractor[effectiveArtifact], error) {
|
||||||
|
return effectiveExtractor{key: "extract"}, nil
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := pipeline.RegisterMerger(catalog.Mergers, pipeline.ModuleSpec{Key: pipeline.DefaultMergeModule, Stage: pipeline.StageMerge, ArtifactKind: effectiveArtifactKind, Requires: []string{"candidate"}, Provides: []string{"merged"}}, func() (contracts.Merger[effectiveArtifact], error) {
|
||||||
|
return effectiveMerger{key: pipeline.DefaultMergeModule}, nil
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := pipeline.RegisterMerger(catalog.Mergers, pipeline.ModuleSpec{Key: "other-merge", Stage: pipeline.StageMerge, ArtifactKind: "other-kind", Requires: []string{"candidate"}, Provides: []string{"merged"}}, func() (contracts.Merger[effectiveArtifact], error) {
|
||||||
|
return effectiveMerger{key: "other-merge"}, nil
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := pipeline.RegisterNormalizer(catalog.Normalizers, pipeline.ModuleSpec{Key: pipeline.DefaultNormalizeModule, Stage: pipeline.StageNormalize, ArtifactKind: effectiveArtifactKind, Requires: []string{"merged"}, Provides: []string{"normalized"}}, func() (contracts.Normalizer[effectiveArtifact], error) {
|
||||||
|
return effectiveNormalizer{key: pipeline.DefaultNormalizeModule}, nil
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := catalog.Outputs.RegisterWithSpec(pipeline.ModuleSpec{Key: pipeline.DefaultOutputModule, Stage: pipeline.StageOutput, Requires: []string{"normalized"}}, func() (contracts.OutputEncoder, error) {
|
||||||
|
return effectiveOutput{key: pipeline.DefaultOutputModule}, nil
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
for _, validator := range []struct {
|
||||||
|
key string
|
||||||
|
class contracts.ExecutionClass
|
||||||
|
}{
|
||||||
|
{key: "default-validator", class: contracts.ExecutionClassDeterministic},
|
||||||
|
{key: "configured-a", class: contracts.ExecutionClassDeterministic},
|
||||||
|
{key: "configured-b", class: contracts.ExecutionClassDeterministic},
|
||||||
|
{key: "llm-validator", class: contracts.ExecutionClassLLMBacked},
|
||||||
|
} {
|
||||||
|
if err := pipeline.RegisterTypedValidatorBuilder(catalog.Validators, effectiveArtifactKind, pipeline.ValidatorSpec{Key: validator.key, ExecutionClass: validator.class}, func(options map[string]any) error {
|
||||||
|
return pipeline.RejectUnknownOptions(options, "nested")
|
||||||
|
}, func(pipeline.BuildRequest) (contracts.TypedValidator[effectiveArtifact], error) {
|
||||||
|
return effectiveValidator{name: validator.key, class: validator.class}, nil
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if err := pipeline.RegisterTypedValidatorBuilder(catalog.Validators, effectiveArtifactKind, pipeline.ValidatorSpec{Key: "option-validator", ExecutionClass: contracts.ExecutionClassDeterministic}, func(options map[string]any) error {
|
||||||
|
return pipeline.RejectUnknownOptions(options, "allowed")
|
||||||
|
}, func(pipeline.BuildRequest) (contracts.TypedValidator[effectiveArtifact], error) {
|
||||||
|
return effectiveValidator{name: "option-validator", class: contracts.ExecutionClassDeterministic}, nil
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := catalog.ValidatorChains.Register(pipeline.ValidatorChainMapping{Stage: pipeline.StageExtract, Module: "extract", Validators: []pipeline.ModuleBinding{pipeline.Binding("default-validator")}}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return catalog
|
||||||
|
}
|
||||||
@@ -1,214 +0,0 @@
|
|||||||
package config
|
|
||||||
|
|
||||||
import (
|
|
||||||
"strings"
|
|
||||||
"testing"
|
|
||||||
"time"
|
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
|
||||||
)
|
|
||||||
|
|
||||||
func TestResolveRejectsEmptyAndUnknownPipelineID(t *testing.T) {
|
|
||||||
tests := []struct {
|
|
||||||
name string
|
|
||||||
pipelineID string
|
|
||||||
want string
|
|
||||||
}{
|
|
||||||
{name: "empty", pipelineID: " ", want: "pipeline id"},
|
|
||||||
{name: "unknown", pipelineID: "missing", want: "not configured"},
|
|
||||||
}
|
|
||||||
|
|
||||||
for _, tc := range tests {
|
|
||||||
t.Run(tc.name, func(t *testing.T) {
|
|
||||||
_, err := validConfig().Resolve(ResolveInput{PipelineID: tc.pipelineID, Catalog: fakeCatalog(t)})
|
|
||||||
if err == nil || !strings.Contains(err.Error(), tc.want) {
|
|
||||||
t.Fatalf("expected error containing %q, got %v", tc.want, err)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestResolveLaneFilteringSuccessAndFailure(t *testing.T) {
|
|
||||||
effective, err := validConfig().Resolve(ResolveInput{
|
|
||||||
PipelineID: " example ",
|
|
||||||
Only: []string{" notes "},
|
|
||||||
Catalog: fakeCatalog(t),
|
|
||||||
})
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("Resolve: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
if effective.PipelineID != "example" {
|
|
||||||
t.Fatalf("unexpected pipeline ID: %q", effective.PipelineID)
|
|
||||||
}
|
|
||||||
if len(effective.ResolvedPipeline.ArtifactLanes) != 1 || effective.ResolvedPipeline.ArtifactLanes[0].ID != "notes" {
|
|
||||||
t.Fatalf("unexpected resolved lanes: %+v", effective.ResolvedPipeline.ArtifactLanes)
|
|
||||||
}
|
|
||||||
if effective.ResolvedPipeline.Digest == "" {
|
|
||||||
t.Fatalf("expected digest")
|
|
||||||
}
|
|
||||||
|
|
||||||
_, err = validConfig().Resolve(ResolveInput{
|
|
||||||
PipelineID: "example",
|
|
||||||
Only: []string{"missing"},
|
|
||||||
Catalog: fakeCatalog(t),
|
|
||||||
})
|
|
||||||
if err == nil || !strings.Contains(err.Error(), "selected artifact lane") {
|
|
||||||
t.Fatalf("expected invalid lane error, got %v", err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestResolveUsesTrimmedPipelineMapKeys(t *testing.T) {
|
|
||||||
cfg := validConfig()
|
|
||||||
cfg.Pipelines[" example "] = cfg.Pipelines["example"]
|
|
||||||
delete(cfg.Pipelines, "example")
|
|
||||||
|
|
||||||
effective, err := cfg.Resolve(ResolveInput{PipelineID: "example", Catalog: fakeCatalog(t)})
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("Resolve: %v", err)
|
|
||||||
}
|
|
||||||
if effective.PipelineID != "example" {
|
|
||||||
t.Fatalf("unexpected pipeline ID: %q", effective.PipelineID)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestResolveSurfacesUnknownModuleKeyThroughCatalog(t *testing.T) {
|
|
||||||
cfg := validConfig()
|
|
||||||
lane := cfg.Pipelines["example"].Artifacts["events"]
|
|
||||||
lane.Extract = pipeline.Binding("missing/extract")
|
|
||||||
cfg.Pipelines["example"].Artifacts["events"] = lane
|
|
||||||
|
|
||||||
_, err := cfg.Resolve(ResolveInput{PipelineID: "example", Catalog: fakeCatalog(t)})
|
|
||||||
if err == nil || !strings.Contains(err.Error(), "missing/extract") || !strings.Contains(err.Error(), "events") {
|
|
||||||
t.Fatalf("expected unknown module error with lane context, got %v", err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestResolveSurfacesMissingCapabilityThroughCatalog(t *testing.T) {
|
|
||||||
_, err := validConfig().Resolve(ResolveInput{
|
|
||||||
PipelineID: "example",
|
|
||||||
Catalog: fakeCatalog(t, pipeline.ModuleSpec{
|
|
||||||
Key: "json",
|
|
||||||
Stage: pipeline.StageOutput,
|
|
||||||
Requires: []string{"missing-capability"},
|
|
||||||
}),
|
|
||||||
})
|
|
||||||
if err == nil || !strings.Contains(err.Error(), "missing capability") || !strings.Contains(err.Error(), "json") {
|
|
||||||
t.Fatalf("expected missing capability error, got %v", err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestResolveDigestChangesWhenEffectiveConfigChanges(t *testing.T) {
|
|
||||||
cfg := validConfig()
|
|
||||||
first, err := cfg.Resolve(ResolveInput{PipelineID: "example", Catalog: fakeCatalog(t)})
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("Resolve first: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
lane := cfg.Pipelines["example"].Artifacts["events"]
|
|
||||||
lane.Extract.Options = map[string]any{"temperature": 0.2}
|
|
||||||
cfg.Pipelines["example"].Artifacts["events"] = lane
|
|
||||||
second, err := cfg.Resolve(ResolveInput{PipelineID: "example", Catalog: fakeCatalog(t)})
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("Resolve second: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
if first.ResolvedPipeline.Digest == second.ResolvedPipeline.Digest {
|
|
||||||
t.Fatalf("expected digest to change, got %q", first.ResolvedPipeline.Digest)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestResolveLLMProfileOverrideAppliesBeforeDigest(t *testing.T) {
|
|
||||||
cfg := validConfig()
|
|
||||||
cfg.LLMProfiles["runtime"] = LLMProfile{Provider: "openai-compatible"}
|
|
||||||
|
|
||||||
base, err := cfg.Resolve(ResolveInput{PipelineID: "example", Catalog: fakeCatalog(t)})
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("Resolve base: %v", err)
|
|
||||||
}
|
|
||||||
effective, err := cfg.Resolve(ResolveInput{
|
|
||||||
PipelineID: "example",
|
|
||||||
Catalog: fakeCatalog(t),
|
|
||||||
LLMProfileOverride: "runtime",
|
|
||||||
})
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("Resolve override: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
if base.ResolvedPipeline.Digest == effective.ResolvedPipeline.Digest {
|
|
||||||
t.Fatalf("expected digest to change after LLM profile override")
|
|
||||||
}
|
|
||||||
for _, binding := range resolvedBindings(effective.ResolvedPipeline) {
|
|
||||||
if binding.LLMProfile != "runtime" {
|
|
||||||
t.Fatalf("binding profile = %q, want runtime", binding.LLMProfile)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
_, err = cfg.Resolve(ResolveInput{
|
|
||||||
PipelineID: "example",
|
|
||||||
Catalog: fakeCatalog(t),
|
|
||||||
LLMProfileOverride: "missing",
|
|
||||||
})
|
|
||||||
if err == nil || !strings.Contains(err.Error(), "LLM profile override") {
|
|
||||||
t.Fatalf("expected override profile error, got %v", err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func resolvedBindings(resolved pipeline.ResolvedPipeline) []pipeline.ModuleBinding {
|
|
||||||
bindings := []pipeline.ModuleBinding{resolved.Input, resolved.Chunk, resolved.Output}
|
|
||||||
for _, lane := range resolved.ArtifactLanes {
|
|
||||||
bindings = append(bindings, lane.Extract, lane.Merge, lane.Normalize)
|
|
||||||
bindings = append(bindings, lane.Validators...)
|
|
||||||
}
|
|
||||||
return bindings
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestOpenAICompatibleClientConfigRejectsIncompleteDefaultProfile(t *testing.T) {
|
|
||||||
cfg := Default()
|
|
||||||
|
|
||||||
_, err := cfg.OpenAICompatibleClientConfig("default")
|
|
||||||
if err == nil || !strings.Contains(err.Error(), "base URL") {
|
|
||||||
t.Fatalf("expected incomplete profile error, got %v", err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestOpenAICompatibleClientConfigSuccess(t *testing.T) {
|
|
||||||
cfg := validConfig()
|
|
||||||
profile := cfg.LLMProfiles["default"]
|
|
||||||
profile.APIKey = "secret"
|
|
||||||
profile.TimeoutSeconds = 45
|
|
||||||
profile.MaxRetries = 4
|
|
||||||
cfg.LLMProfiles["default"] = profile
|
|
||||||
|
|
||||||
llmCfg, err := cfg.OpenAICompatibleClientConfig(" default ")
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("OpenAICompatibleClientConfig: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
if llmCfg.BaseURL != "https://example.invalid/v1" || llmCfg.Model != "test-model" || llmCfg.APIKey != "secret" {
|
|
||||||
t.Fatalf("unexpected client config strings: %+v", llmCfg)
|
|
||||||
}
|
|
||||||
if llmCfg.MaxRetries != 4 {
|
|
||||||
t.Fatalf("unexpected max retries: %d", llmCfg.MaxRetries)
|
|
||||||
}
|
|
||||||
if llmCfg.RequestTimeout != 45*time.Second {
|
|
||||||
t.Fatalf("unexpected timeout: %s", llmCfg.RequestTimeout)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestOpenAICompatibleClientConfigRejectsUnknownAndUnsupportedProfiles(t *testing.T) {
|
|
||||||
_, err := validConfig().OpenAICompatibleClientConfig("missing")
|
|
||||||
if err == nil || !strings.Contains(err.Error(), "not configured") {
|
|
||||||
t.Fatalf("expected unknown profile error, got %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
cfg := validConfig()
|
|
||||||
profile := cfg.LLMProfiles["default"]
|
|
||||||
profile.Provider = "unsupported"
|
|
||||||
cfg.LLMProfiles["default"] = profile
|
|
||||||
|
|
||||||
_, err = cfg.OpenAICompatibleClientConfig("default")
|
|
||||||
if err == nil || !strings.Contains(err.Error(), "provider") {
|
|
||||||
t.Fatalf("expected unsupported provider error, got %v", err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -6,7 +6,6 @@ import (
|
|||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/core/diagnostics"
|
|
||||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -30,43 +29,6 @@ func (c *Config) applyEnvOverridesWithLookup(lookup func(string) (string, bool))
|
|||||||
if c == nil {
|
if c == nil {
|
||||||
return fmt.Errorf("config must not be nil")
|
return fmt.Errorf("config must not be nil")
|
||||||
}
|
}
|
||||||
if c.LLMProfiles == nil {
|
|
||||||
c.LLMProfiles = map[string]LLMProfile{}
|
|
||||||
}
|
|
||||||
|
|
||||||
defaultProfile := c.LLMProfiles[pipeline.DefaultLLMProfile]
|
|
||||||
if raw, ok := lookup("NOTARIUS_LLM_DEFAULT_API_KEY"); ok {
|
|
||||||
defaultProfile.APIKey = raw
|
|
||||||
}
|
|
||||||
if raw, ok := lookup("NOTARIUS_LLM_DEFAULT_BASE_URL"); ok {
|
|
||||||
defaultProfile.BaseURL = strings.TrimSpace(raw)
|
|
||||||
}
|
|
||||||
if raw, ok := lookup("NOTARIUS_LLM_DEFAULT_MODEL"); ok {
|
|
||||||
defaultProfile.Model = strings.TrimSpace(raw)
|
|
||||||
}
|
|
||||||
if raw, ok := lookup("NOTARIUS_LLM_DEFAULT_TIMEOUT_SECONDS"); ok {
|
|
||||||
value, err := parseIntEnv("NOTARIUS_LLM_DEFAULT_TIMEOUT_SECONDS", raw)
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
defaultProfile.TimeoutSeconds = value
|
|
||||||
}
|
|
||||||
if raw, ok := lookup("NOTARIUS_LLM_DEFAULT_MAX_RETRIES"); ok {
|
|
||||||
value, err := parseIntEnv("NOTARIUS_LLM_DEFAULT_MAX_RETRIES", raw)
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
defaultProfile.MaxRetries = value
|
|
||||||
}
|
|
||||||
if raw, ok := lookup("NOTARIUS_LLM_DEFAULT_MAX_CONCURRENCY"); ok {
|
|
||||||
value, err := parseIntEnv("NOTARIUS_LLM_DEFAULT_MAX_CONCURRENCY", raw)
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
defaultProfile.MaxConcurrency = value
|
|
||||||
}
|
|
||||||
c.LLMProfiles[pipeline.DefaultLLMProfile] = defaultProfile
|
|
||||||
|
|
||||||
if raw, ok := lookup("NOTARIUS_TOTAL_LLM_CONCURRENCY"); ok {
|
if raw, ok := lookup("NOTARIUS_TOTAL_LLM_CONCURRENCY"); ok {
|
||||||
value, err := parseIntEnv("NOTARIUS_TOTAL_LLM_CONCURRENCY", raw)
|
value, err := parseIntEnv("NOTARIUS_TOTAL_LLM_CONCURRENCY", raw)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -74,11 +36,60 @@ func (c *Config) applyEnvOverridesWithLookup(lookup func(string) (string, bool))
|
|||||||
}
|
}
|
||||||
c.Concurrency.TotalLLM = value
|
c.Concurrency.TotalLLM = value
|
||||||
}
|
}
|
||||||
if raw, ok := lookup("NOTARIUS_WORK_DIR"); ok {
|
if raw, ok := lookup("NOTARIUS_STAGE_WORKERS_EXTRACT"); ok {
|
||||||
c.Diagnostics.WorkDir = strings.TrimSpace(raw)
|
value, err := parseIntEnv("NOTARIUS_STAGE_WORKERS_EXTRACT", raw)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if c.Concurrency.StageWorkers == nil {
|
||||||
|
c.Concurrency.StageWorkers = make(map[string]int)
|
||||||
|
}
|
||||||
|
c.Concurrency.StageWorkers["extract"] = value
|
||||||
|
c.Concurrency.extractWorkersConfigured = true
|
||||||
|
}
|
||||||
|
c.Concurrency.recomputeStageWorkerDefaults()
|
||||||
|
if raw, ok := lookup("NOTARIUS_OUTPUT_DIR"); ok {
|
||||||
|
c.Output.Directory = strings.TrimSpace(raw)
|
||||||
|
if c.Output.Directory == "" {
|
||||||
|
return fmt.Errorf("NOTARIUS_OUTPUT_DIR: must not be empty")
|
||||||
|
}
|
||||||
|
if strings.ContainsRune(c.Output.Directory, '\x00') {
|
||||||
|
return fmt.Errorf("NOTARIUS_OUTPUT_DIR: must not contain NUL")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if raw, ok := lookup("NOTARIUS_CACHE_CHUNK_PLANS_MODE"); ok {
|
||||||
|
mode, err := pipeline.ParseChunkCacheMode(raw)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("NOTARIUS_CACHE_CHUNK_PLANS_MODE: %w", err)
|
||||||
|
}
|
||||||
|
c.Cache.ChunkPlans.Mode = mode
|
||||||
|
}
|
||||||
|
if raw, ok := lookup("NOTARIUS_CACHE_CHUNK_PLANS_DIR"); ok {
|
||||||
|
c.Cache.ChunkPlans.Directory = cleanOptionalPath(raw)
|
||||||
|
if c.Cache.ChunkPlans.Directory == "" {
|
||||||
|
return fmt.Errorf("NOTARIUS_CACHE_CHUNK_PLANS_DIR: must not be empty")
|
||||||
|
}
|
||||||
|
if strings.ContainsRune(c.Cache.ChunkPlans.Directory, '\x00') {
|
||||||
|
return fmt.Errorf("NOTARIUS_CACHE_CHUNK_PLANS_DIR: must not contain NUL")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if raw, ok := lookup("NOTARIUS_CACHE_CHECKPOINTS_DIR"); ok {
|
||||||
|
c.Cache.Checkpoints.Directory = cleanOptionalPath(raw)
|
||||||
|
if c.Cache.Checkpoints.Directory == "" {
|
||||||
|
return fmt.Errorf("NOTARIUS_CACHE_CHECKPOINTS_DIR: must not be empty")
|
||||||
|
}
|
||||||
|
if strings.ContainsRune(c.Cache.Checkpoints.Directory, '\x00') {
|
||||||
|
return fmt.Errorf("NOTARIUS_CACHE_CHECKPOINTS_DIR: must not contain NUL")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if raw, ok := lookup("NOTARIUS_DEBUG_DIR"); ok {
|
||||||
|
c.Debug.Directory = strings.TrimSpace(raw)
|
||||||
|
if c.Debug.Directory == "" {
|
||||||
|
return fmt.Errorf("NOTARIUS_DEBUG_DIR: must not be empty")
|
||||||
|
}
|
||||||
|
if strings.ContainsRune(c.Debug.Directory, '\x00') {
|
||||||
|
return fmt.Errorf("NOTARIUS_DEBUG_DIR: must not contain NUL")
|
||||||
}
|
}
|
||||||
if raw, ok := lookup("NOTARIUS_DIAGNOSTICS_RETENTION"); ok {
|
|
||||||
c.Diagnostics.Retention = diagnostics.RetentionMode(strings.TrimSpace(raw))
|
|
||||||
}
|
}
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|||||||
239
internal/core/config/env_contract_test.go
Normal file
239
internal/core/config/env_contract_test.go
Normal file
@@ -0,0 +1,239 @@
|
|||||||
|
package config
|
||||||
|
|
||||||
|
import (
|
||||||
|
"errors"
|
||||||
|
"reflect"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestPrecedenceFileValuesOverrideBuiltInDefaults(t *testing.T) {
|
||||||
|
cfg := applyFileConfig(t, `version: 3
|
||||||
|
concurrency:
|
||||||
|
total_llm: 4
|
||||||
|
stage_workers:
|
||||||
|
extract: 2
|
||||||
|
output:
|
||||||
|
directory: ./file-output
|
||||||
|
cache:
|
||||||
|
chunk_plans:
|
||||||
|
directory: ./file-plans
|
||||||
|
mode: refresh
|
||||||
|
checkpoints:
|
||||||
|
directory: ./file-checkpoints
|
||||||
|
debug:
|
||||||
|
directory: ./file-debug
|
||||||
|
`)
|
||||||
|
if cfg.Concurrency.TotalLLM != 4 || cfg.Concurrency.StageWorkers["extract"] != 2 ||
|
||||||
|
cfg.Output.Directory != "./file-output" || cfg.Cache.ChunkPlans.Directory != "file-plans" ||
|
||||||
|
cfg.Cache.ChunkPlans.Mode != pipeline.ChunkCacheRefresh || cfg.Cache.Checkpoints.Directory != "file-checkpoints" ||
|
||||||
|
cfg.Debug.Directory != "./file-debug" {
|
||||||
|
t.Fatalf("file values did not override defaults: %#v", cfg)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestPrecedenceOperationalEnvironmentOverridesFileValues(t *testing.T) {
|
||||||
|
cfg := applyFileConfig(t, `version: 3
|
||||||
|
concurrency:
|
||||||
|
total_llm: 2
|
||||||
|
stage_workers:
|
||||||
|
extract: 1
|
||||||
|
output:
|
||||||
|
directory: ./file-output
|
||||||
|
cache:
|
||||||
|
chunk_plans:
|
||||||
|
directory: ./file-plans
|
||||||
|
mode: refresh
|
||||||
|
checkpoints:
|
||||||
|
directory: ./file-checkpoints
|
||||||
|
debug:
|
||||||
|
directory: ./file-debug
|
||||||
|
`)
|
||||||
|
env := map[string]string{
|
||||||
|
"NOTARIUS_TOTAL_LLM_CONCURRENCY": "8",
|
||||||
|
"NOTARIUS_STAGE_WORKERS_EXTRACT": "6",
|
||||||
|
"NOTARIUS_OUTPUT_DIR": "/env/output",
|
||||||
|
"NOTARIUS_CACHE_CHUNK_PLANS_MODE": "bypass",
|
||||||
|
"NOTARIUS_CACHE_CHUNK_PLANS_DIR": "/env/plans",
|
||||||
|
"NOTARIUS_CACHE_CHECKPOINTS_DIR": "/env/checkpoints",
|
||||||
|
"NOTARIUS_DEBUG_DIR": "/env/debug",
|
||||||
|
}
|
||||||
|
if err := cfg.ApplyEnvOverridesWithLookup(lookupValues(env)); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if cfg.Concurrency.TotalLLM != 8 || cfg.Concurrency.StageWorkers["extract"] != 6 ||
|
||||||
|
cfg.Output.Directory != "/env/output" || cfg.Cache.ChunkPlans.Directory != "/env/plans" ||
|
||||||
|
cfg.Cache.ChunkPlans.Mode != pipeline.ChunkCacheBypass || cfg.Cache.Checkpoints.Directory != "/env/checkpoints" ||
|
||||||
|
cfg.Debug.Directory != "/env/debug" {
|
||||||
|
t.Fatalf("environment values did not override file values: %#v", cfg)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestPrecedenceExtractWorkersFollowEffectiveConcurrencyUnlessExplicit(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
file string
|
||||||
|
env map[string]string
|
||||||
|
wantTotal int
|
||||||
|
wantWorker int
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
name: "default follows environment total",
|
||||||
|
file: "version: 3\n",
|
||||||
|
env: map[string]string{"NOTARIUS_TOTAL_LLM_CONCURRENCY": "5"},
|
||||||
|
wantTotal: 5,
|
||||||
|
wantWorker: 5,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "file worker is retained",
|
||||||
|
file: "version: 3\nconcurrency:\n total_llm: 3\n stage_workers:\n extract: 2\n",
|
||||||
|
env: map[string]string{"NOTARIUS_TOTAL_LLM_CONCURRENCY": "6"},
|
||||||
|
wantTotal: 6,
|
||||||
|
wantWorker: 2,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "environment worker is retained",
|
||||||
|
file: "version: 3\nconcurrency:\n total_llm: 2\n",
|
||||||
|
env: map[string]string{
|
||||||
|
"NOTARIUS_TOTAL_LLM_CONCURRENCY": "6",
|
||||||
|
"NOTARIUS_STAGE_WORKERS_EXTRACT": "4",
|
||||||
|
},
|
||||||
|
wantTotal: 6,
|
||||||
|
wantWorker: 4,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
cfg := applyFileConfig(t, tt.file)
|
||||||
|
if err := cfg.ApplyEnvOverridesWithLookup(lookupValues(tt.env)); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if cfg.Concurrency.TotalLLM != tt.wantTotal || cfg.Concurrency.StageWorkers["extract"] != tt.wantWorker {
|
||||||
|
t.Fatalf("concurrency = %#v, want total %d and extract %d", cfg.Concurrency, tt.wantTotal, tt.wantWorker)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestPrecedenceEmptyFileCacheDirectoriesDeferPerUserResolution(t *testing.T) {
|
||||||
|
cfg := applyFileConfig(t, `version: 3
|
||||||
|
cache:
|
||||||
|
chunk_plans:
|
||||||
|
directory: ""
|
||||||
|
checkpoints:
|
||||||
|
directory: ""
|
||||||
|
`)
|
||||||
|
if err := cfg.Validate(); err != nil {
|
||||||
|
t.Fatalf("empty file cache directories should be valid: %v", err)
|
||||||
|
}
|
||||||
|
if cfg.Cache.ChunkPlans.Directory != "" || cfg.Cache.Checkpoints.Directory != "" {
|
||||||
|
t.Fatalf("empty cache directories were not preserved for deferred resolution: %#v", cfg.Cache)
|
||||||
|
}
|
||||||
|
resolver := func() (string, error) { return "/user/cache", nil }
|
||||||
|
chunkPlans, err := DefaultChunkPlanRoot(resolver)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
checkpoints, err := DefaultCheckpointRoot(resolver)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if chunkPlans != "/user/cache/notarius/chunk-plans" || checkpoints != "/user/cache/notarius/checkpoints" {
|
||||||
|
t.Fatalf("deferred cache roots = %q, %q", chunkPlans, checkpoints)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDefaultCacheRootsRejectInvalidUserCacheResolvers(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
resolver func() (string, error)
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{name: "nil resolver", want: "must not be nil"},
|
||||||
|
{
|
||||||
|
name: "resolver failure",
|
||||||
|
resolver: func() (string, error) {
|
||||||
|
return "", errors.New("cache home unavailable")
|
||||||
|
},
|
||||||
|
want: "resolve user cache directory",
|
||||||
|
},
|
||||||
|
{name: "empty directory", resolver: func() (string, error) { return " ", nil }, want: "must not be empty"},
|
||||||
|
}
|
||||||
|
families := []struct {
|
||||||
|
name string
|
||||||
|
root func(func() (string, error)) (string, error)
|
||||||
|
}{
|
||||||
|
{name: "chunk plans", root: DefaultChunkPlanRoot},
|
||||||
|
{name: "checkpoints", root: DefaultCheckpointRoot},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, family := range families {
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(family.name+"/"+tt.name, func(t *testing.T) {
|
||||||
|
_, err := family.root(tt.resolver)
|
||||||
|
if err == nil || !strings.Contains(err.Error(), tt.want) {
|
||||||
|
t.Fatalf("error = %v, want substring %q", err, tt.want)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEnvEmptyDirectoryOverridesAreErrors(t *testing.T) {
|
||||||
|
tests := []string{
|
||||||
|
"NOTARIUS_OUTPUT_DIR",
|
||||||
|
"NOTARIUS_CACHE_CHUNK_PLANS_DIR",
|
||||||
|
"NOTARIUS_CACHE_CHECKPOINTS_DIR",
|
||||||
|
"NOTARIUS_DEBUG_DIR",
|
||||||
|
}
|
||||||
|
for _, name := range tests {
|
||||||
|
t.Run(name, func(t *testing.T) {
|
||||||
|
cfg := Default()
|
||||||
|
err := cfg.ApplyEnvOverridesWithLookup(lookupValues(map[string]string{name: " \t"}))
|
||||||
|
if err == nil || !strings.Contains(err.Error(), name) {
|
||||||
|
t.Fatalf("error = %v, want responsible environment variable", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEnvInvalidIntegersAndChunkCacheModesReportTheirNames(t *testing.T) {
|
||||||
|
tests := map[string]string{
|
||||||
|
"NOTARIUS_TOTAL_LLM_CONCURRENCY": "not-an-integer",
|
||||||
|
"NOTARIUS_STAGE_WORKERS_EXTRACT": "not-an-integer",
|
||||||
|
"NOTARIUS_CACHE_CHUNK_PLANS_MODE": "not-a-cache-mode",
|
||||||
|
}
|
||||||
|
for name, value := range tests {
|
||||||
|
t.Run(name, func(t *testing.T) {
|
||||||
|
cfg := Default()
|
||||||
|
err := cfg.ApplyEnvOverridesWithLookup(lookupValues(map[string]string{name: value}))
|
||||||
|
if err == nil || !strings.Contains(err.Error(), name) {
|
||||||
|
t.Fatalf("error = %v, want responsible environment variable", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEnvRemovedProviderVariablesAreIgnored(t *testing.T) {
|
||||||
|
before := Default()
|
||||||
|
cfg := Default()
|
||||||
|
removed := map[string]string{
|
||||||
|
"NOTARIUS_LLM_DEFAULT_ENDPOINT": "ignored-provider-setting",
|
||||||
|
"NOTARIUS_LLM_DEFAULT_MODEL": "ignored-provider-setting",
|
||||||
|
}
|
||||||
|
if err := cfg.ApplyEnvOverridesWithLookup(lookupValues(removed)); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if !reflect.DeepEqual(cfg, before) {
|
||||||
|
t.Fatalf("removed provider variables changed configuration: %#v", cfg)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func lookupValues(values map[string]string) func(string) (string, bool) {
|
||||||
|
return func(name string) (string, bool) {
|
||||||
|
value, ok := values[name]
|
||||||
|
return value, ok
|
||||||
|
}
|
||||||
|
}
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user