Compare commits
237 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 545aa6893b | |||
| 98139f7e8b | |||
| f8fa0a2623 | |||
| 9af773491b | |||
| 2a656f0f11 | |||
| aec35a2d9b | |||
| 23dc4e2078 | |||
| c812fe3655 | |||
| 3da97ca50c | |||
| 4c203d8588 | |||
| fcb5f825e1 | |||
| 4c57ace2f6 | |||
| dde7f76ecb | |||
| a102db36af | |||
| b5b1d22011 | |||
| c91599ef36 | |||
| 257f10c9fb | |||
| fb9a4d14f4 | |||
| c4435b76c4 | |||
| 61000a9466 | |||
| 7e4ceb3d48 | |||
| 4e991fa21d | |||
| f86b17045d | |||
| f302488075 | |||
| 8c1171478d | |||
| 7ee637803d | |||
| 4d6086fefb | |||
| 82cb53e107 | |||
| b97b12da7f | |||
| 49ea747b17 | |||
| 3b5e9db41f | |||
| 8b4b328c4e | |||
| ee2b8e63e6 | |||
| 51edd384c0 | |||
| b804d0f2c8 | |||
| fd5ccc668b | |||
| 0dc8ff9b52 | |||
| 8657a28bdb | |||
| a9c5e4ad4e | |||
| 2176b4371d | |||
| effc10d75b | |||
| 5887839aa1 | |||
| 3128bef20a | |||
| 4e4e2b7d96 | |||
| 99f4f9a0db | |||
| 6abdd67bb5 | |||
| c32e0c401f | |||
| ab5751459a | |||
| 62de6abdbf | |||
| 903dc70682 | |||
| 23c714da66 | |||
| 6639775d7d | |||
| 966b95b176 | |||
| 3bcf2c08dd | |||
| 700ab655ca | |||
| 85c5647385 | |||
| 2ef7c76d99 | |||
| 9bc1b0feda | |||
| 5cec84a4a7 | |||
| abfbe42d61 | |||
| e7e3bef1e4 | |||
| a68e8e31a4 | |||
| 3a9e60cda9 | |||
| 905ff03ccc | |||
| 495f7bcde4 | |||
| 51e0e8c5d0 | |||
| a2409a1fd1 | |||
| b3363f87d6 | |||
| 5831c0c9e6 | |||
| 42ed81cbe1 | |||
| e433c86203 | |||
| a2a144dffa | |||
| 8ef6e99d69 | |||
| 2545faef6c | |||
| 80be8be4d6 | |||
| 801adb385d | |||
| feba7b9d74 | |||
| b89224bbde | |||
| 131ffd9887 | |||
| af492c9e97 | |||
| f39fc94610 | |||
| 4e4eff6ba7 | |||
| 8ff1b4fa66 | |||
| 702f622e18 | |||
| 9da2c1e144 | |||
| 32653f54f9 | |||
| 72a200968a | |||
| b39b68add7 | |||
| d9fa1d9328 | |||
| 8375ad83f3 | |||
| 4158394dcf | |||
| eac7e155a5 | |||
| 0cf2cbfeb3 | |||
| 361dbb4ca8 | |||
| d6deccf3e8 | |||
| ee747243fe | |||
| a1ceb457e9 | |||
| 9900211fa4 | |||
| 60cebf0e4b | |||
| 7bd575187e | |||
| ab5a7e8e3d | |||
| 99b2e1cd81 | |||
| 363313d99c | |||
| 18ddf00d3d | |||
| 59f3fe3d1d | |||
| 1dccf5f140 | |||
| 0b40cf8026 | |||
| a7ec195587 | |||
| 13de820931 | |||
| 14ef59aaed | |||
| f387222fce | |||
| 9cb9008dfc | |||
| 083decc5b4 | |||
| 57cac5d3f7 | |||
| 0a772e03b4 | |||
| 0920062a38 | |||
| 39afe644eb | |||
| e3ee3de10a | |||
| 9c72db56e9 | |||
| bb2d606dbb | |||
| 9850767a8a | |||
| 74e2d21de5 | |||
| 7cb18a1a40 | |||
| b556fc2f4f | |||
| b99bd38eb4 | |||
| 701b6726d7 | |||
| 665039f4dc | |||
| ef8dae776e | |||
| d01775b68a | |||
| 0d6f2dd0ce | |||
| df40cbec6e | |||
| 0341e0c7c0 | |||
| 39af7d4f3c | |||
| bba582b4ca | |||
| 1f16a85330 | |||
| f9482639d4 | |||
| dce721cdbd | |||
| 98734644d6 | |||
| 951383226c | |||
| df58595d1e | |||
| c3c14e7468 | |||
| e7319ea016 | |||
| bd2d5e2496 | |||
| 115a44f629 | |||
| 18411dc5b5 | |||
| e23dc1ab6e | |||
| e1359ea227 | |||
| 7fdd99ec27 | |||
| a90231ce0c | |||
| ed879b8bb0 | |||
| 717451512a | |||
| 3ddb3a947b | |||
| c6632d5576 | |||
| ffc07922c7 | |||
| f3310d4d16 | |||
| 88cee96d8d | |||
| 2fece10215 | |||
| 0658f2f642 | |||
| a51228c803 | |||
| 4491fb5ccd | |||
| 30b905765c | |||
| 03eac70881 | |||
| 0f7e6b979f | |||
| c366912586 | |||
| 9fe44cd00d | |||
| 094b0d2532 | |||
| 98649f4d81 | |||
| 8a559efd5b | |||
| 72deccb4e2 | |||
| 5620fc5bcf | |||
| be57e675e0 | |||
| 3971443831 | |||
| a6b0c33e9f | |||
| 96b886e711 | |||
| 7d584ee6cd | |||
| 572a112c31 | |||
| ea87c335d6 | |||
| 7169ff04df | |||
| ef1f650bc0 | |||
| 0d02cb9fa0 | |||
| 0299b128cf | |||
| d723384888 | |||
| 54228055c8 | |||
| 23ed716450 | |||
| ab59bab044 | |||
| 71395bb076 | |||
| 79737edf79 | |||
| df2c765b7f | |||
| f050b9dd54 | |||
| 9c9cb54339 | |||
| 7657ec3ad6 | |||
| cee52aa092 | |||
| e920f3a8d5 | |||
| 591c529a09 | |||
| 7324c5a686 | |||
| d0936fb022 | |||
| 2aa074c5cf | |||
| 782d0cf3b9 | |||
| 083c01cfa0 | |||
| 2937696024 | |||
| b817a5b772 | |||
| 3022f20beb | |||
| ca1ded1821 | |||
| 3752f3ed28 | |||
| 870c2d69d5 | |||
| 135407ba7c | |||
| 228c348e42 | |||
| a813bd5a50 | |||
| d8f58dce31 | |||
| 7111edeca4 | |||
| 3aae4bbb12 | |||
| b29d8eeb50 | |||
| dffb432537 | |||
| 2dd38c7913 | |||
| bc2ade38d9 | |||
| 5be831eb13 | |||
| cae4d99a89 | |||
| e09dc0512d | |||
| ae82bc1ce0 | |||
| 01eb7aa1aa | |||
| 2ca700195c | |||
| 2b08c34539 | |||
| 79f1fc1e09 | |||
| 9c753270bd | |||
| b907cb01aa | |||
| 7824afd4a5 | |||
| 2a4e1e912c | |||
| dd03c09d75 | |||
| 5bc8e8683f | |||
| 648001a8fe | |||
| 6684774f52 | |||
| f3b63bd5e5 | |||
| 23d6470b0f | |||
| 128449040f | |||
| 02ab106ade | |||
| c128970f58 | |||
| d001baa660 |
@@ -2,39 +2,26 @@ when:
|
|||||||
- event: tag
|
- event: tag
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- name: build-release-assets
|
validate-release:
|
||||||
image: golang:1.25
|
image: golang:1.25.5
|
||||||
|
commands:
|
||||||
|
- ./scripts/check-release-candidate.sh "$CI_COMMIT_TAG"
|
||||||
|
|
||||||
|
build-release-assets:
|
||||||
|
image: golang:1.25.5
|
||||||
|
depends_on:
|
||||||
|
- validate-release
|
||||||
commands:
|
commands:
|
||||||
- |
|
- |
|
||||||
set -eu
|
set -eu
|
||||||
|
case "$PWD" in
|
||||||
|
/*) ;;
|
||||||
|
*) echo "release workspace must have an absolute path" >&2; exit 1 ;;
|
||||||
|
esac
|
||||||
|
./scripts/build-release-assets.sh "$CI_COMMIT_TAG" "$PWD/dist"
|
||||||
|
|
||||||
version="$CI_COMMIT_TAG"
|
publish-release:
|
||||||
dist="dist"
|
image: woodpeckerci/plugin-release:0.3.1
|
||||||
pkg="gitea.maximumdirect.net/eric/narratio/cmd/narratio"
|
|
||||||
|
|
||||||
rm -rf "$dist"
|
|
||||||
mkdir -p "$dist"
|
|
||||||
|
|
||||||
build_binary() {
|
|
||||||
goos="$1"
|
|
||||||
goarch="$2"
|
|
||||||
suffix="$3"
|
|
||||||
output="$dist/narratio-$version-$goos-$goarch$suffix"
|
|
||||||
|
|
||||||
CGO_ENABLED=0 GOOS="$goos" GOARCH="$goarch" \
|
|
||||||
go build -trimpath -ldflags "-s -w -X gitea.maximumdirect.net/eric/narratio/internal/buildinfo.Version=$version" \
|
|
||||||
-o "$output" "$pkg"
|
|
||||||
}
|
|
||||||
|
|
||||||
build_binary linux amd64 ""
|
|
||||||
build_binary linux arm64 ""
|
|
||||||
build_binary darwin amd64 ""
|
|
||||||
build_binary darwin arm64 ""
|
|
||||||
build_binary windows amd64 ".exe"
|
|
||||||
build_binary windows arm64 ".exe"
|
|
||||||
|
|
||||||
- name: publish-release
|
|
||||||
image: woodpeckerci/plugin-release
|
|
||||||
depends_on:
|
depends_on:
|
||||||
- build-release-assets
|
- build-release-assets
|
||||||
settings:
|
settings:
|
||||||
@@ -42,6 +29,8 @@ steps:
|
|||||||
from_secret: GITEA_RELEASE_TOKEN
|
from_secret: GITEA_RELEASE_TOKEN
|
||||||
files:
|
files:
|
||||||
- dist/narratio-*
|
- dist/narratio-*
|
||||||
|
title: Narratio ${CI_COMMIT_TAG}
|
||||||
|
note: docs/releases/${CI_COMMIT_TAG}.md
|
||||||
checksum: sha256
|
checksum: sha256
|
||||||
checksum-file: SHA256SUMS
|
checksum-file: SHA256SUMS
|
||||||
checksum-flatten: true
|
checksum-flatten: true
|
||||||
|
|||||||
8
.woodpecker/shuffle.yml
Normal file
8
.woodpecker/shuffle.yml
Normal file
@@ -0,0 +1,8 @@
|
|||||||
|
when:
|
||||||
|
- event: cron
|
||||||
|
|
||||||
|
steps:
|
||||||
|
shuffled-race-tests:
|
||||||
|
image: golang:1.25
|
||||||
|
commands:
|
||||||
|
- go test -race -shuffle=on -count=3 ./...
|
||||||
47
.woodpecker/verify.yml
Normal file
47
.woodpecker/verify.yml
Normal file
@@ -0,0 +1,47 @@
|
|||||||
|
when:
|
||||||
|
- event: [push, pull_request]
|
||||||
|
|
||||||
|
steps:
|
||||||
|
tests:
|
||||||
|
image: golang:1.25
|
||||||
|
commands:
|
||||||
|
- go test ./...
|
||||||
|
|
||||||
|
race-tests:
|
||||||
|
image: golang:1.25
|
||||||
|
depends_on: tests
|
||||||
|
commands:
|
||||||
|
- go test -race ./...
|
||||||
|
|
||||||
|
static-analysis:
|
||||||
|
image: golang:1.25
|
||||||
|
depends_on: tests
|
||||||
|
commands:
|
||||||
|
- go vet ./...
|
||||||
|
|
||||||
|
build:
|
||||||
|
image: golang:1.25
|
||||||
|
depends_on: tests
|
||||||
|
commands:
|
||||||
|
- go build ./...
|
||||||
|
|
||||||
|
documentation-and-examples:
|
||||||
|
image: golang:1.25
|
||||||
|
depends_on: tests
|
||||||
|
commands:
|
||||||
|
- go test ./internal/doccheck
|
||||||
|
- go test ./internal/config -run '^TestExamplesLoadAndValidate$'
|
||||||
|
|
||||||
|
cross-build:
|
||||||
|
image: golang:1.25
|
||||||
|
depends_on: [race-tests, static-analysis, build, documentation-and-examples]
|
||||||
|
commands:
|
||||||
|
- |
|
||||||
|
set -eu
|
||||||
|
output_dir="$(mktemp -d)"
|
||||||
|
trap 'rm -rf "$output_dir"' EXIT
|
||||||
|
for target in linux/amd64 linux/arm64 darwin/amd64 darwin/arm64 windows/amd64 windows/arm64; do
|
||||||
|
goos="${target%/*}"
|
||||||
|
goarch="${target#*/}"
|
||||||
|
CGO_ENABLED=0 GOOS="$goos" GOARCH="$goarch" go build -o "$output_dir/narratio-$goos-$goarch" ./cmd/narratio
|
||||||
|
done
|
||||||
46
README.md
46
README.md
@@ -1,22 +1,42 @@
|
|||||||
# narratio
|
# narratio
|
||||||
|
|
||||||
Narratio is a Go orchestration application that turns D&D session audio into polished transcripts and generated session artifacts.
|
Narratio is a stage-driven Go orchestrator for turning D&D session audio into
|
||||||
|
polished transcripts, validated Notarius extraction lanes, and generated
|
||||||
|
artifacts.
|
||||||
|
|
||||||
It coordinates transcription, merge/polish/normalize/trim processing, artifact generation, archive publishing, and resumable run state in one operator workflow.
|
It runs a deterministic workflow with manifest-driven continuation, remote
|
||||||
|
publish, and restore support.
|
||||||
|
|
||||||
```bash
|
```sh
|
||||||
narratio run --session-id 2026-04-04
|
narratio run 2026-04-04
|
||||||
```
|
```
|
||||||
|
|
||||||
This command requires discoverable `pipeline.yml` and `session.yml` files (or explicit `--config` and `--session` flags).
|
This requires resolvable `pipeline.yml`, `campaign.yml`, and concrete
|
||||||
|
`session.yml` files or their explicit command-line alternatives.
|
||||||
|
|
||||||
## Documentation
|
## Documentation
|
||||||
|
|
||||||
- [Configuration](docs/config.md)
|
- [CLI reference](docs/cli.md) — commands, arguments, flags, and invocation
|
||||||
- [CLI Reference](docs/cli.md)
|
behavior.
|
||||||
- [Operations and Recovery](docs/operations.md)
|
- [Configuration](docs/config.md) — discovery, fields, defaults, and
|
||||||
- [Troubleshooting](docs/troubleshooting.md)
|
validation.
|
||||||
- [Development Guide](docs/development.md)
|
- [Operations](docs/operations.md) — runtime workflow, state, publishing,
|
||||||
- [Architecture Principles](docs/architecture.md)
|
recovery, and cleanup.
|
||||||
- [Internal Component Contracts](docs/internal/README.md)
|
- [Troubleshooting](docs/troubleshooting.md) — symptom-driven diagnosis and
|
||||||
- [Config Examples](examples/)
|
safe remedies.
|
||||||
|
- [Integration contracts](docs/integrations/) — external tools, formats, and
|
||||||
|
compatibility expectations.
|
||||||
|
- [Maintained examples](examples/README.md) — complete copyable configuration
|
||||||
|
and input files, including the production/testing split bundle.
|
||||||
|
|
||||||
|
## Maintainer Documentation
|
||||||
|
|
||||||
|
- [Development guide](docs/development.md) — first-read orientation and
|
||||||
|
task-specific reading routes.
|
||||||
|
- [Internal overview](docs/internal/overview.md) — implemented component map.
|
||||||
|
- [Architecture](docs/policy/architecture.md) — normative boundaries and
|
||||||
|
invariants.
|
||||||
|
- [Documentation policy](docs/policy/documentation.md) — canonical ownership
|
||||||
|
and maintenance rules.
|
||||||
|
- [Testing policy](docs/policy/testing.md) — test value, boundaries, and
|
||||||
|
sufficiency.
|
||||||
|
|||||||
@@ -1,202 +0,0 @@
|
|||||||
# Narratio Architecture
|
|
||||||
|
|
||||||
## Purpose
|
|
||||||
|
|
||||||
`narratio` is a Go orchestration application for processing D&D session audio into polished transcripts and generated session artifacts.
|
|
||||||
|
|
||||||
This document defines the development principles for the project. It is inward-facing: its audience is developers and LLM coding agents. It should guide future changes, not serve as a complete implementation reference.
|
|
||||||
|
|
||||||
Implemented component details belong under `docs/internal/`.
|
|
||||||
|
|
||||||
## Project Shape
|
|
||||||
|
|
||||||
Narratio is a modular, stage-driven orchestrator.
|
|
||||||
|
|
||||||
It coordinates specialized downstream systems rather than reimplementing their domains:
|
|
||||||
|
|
||||||
- WhisperX handles transcription.
|
|
||||||
- Seriatim handles deterministic transcript merge/normalization/trim behavior.
|
|
||||||
- Audita handles transcript correction and polishing.
|
|
||||||
- Scriptorium handles prompt execution and generated artifacts.
|
|
||||||
|
|
||||||
Narratio owns orchestration, configuration loading, session/run state, local and remote path modeling, manifest persistence, stage sequencing, resume behavior, and archive semantics.
|
|
||||||
|
|
||||||
Narratio should remain explicit and comprehensible. It is not intended to become a generic workflow engine.
|
|
||||||
|
|
||||||
## Core Principles
|
|
||||||
|
|
||||||
### Modular and composable
|
|
||||||
|
|
||||||
Code should be organized around clear responsibilities. Stages, adapters, config loading, manifest persistence, path construction, and storage behavior should remain separable and independently testable.
|
|
||||||
|
|
||||||
### Hexagonal boundaries
|
|
||||||
|
|
||||||
External systems should be isolated behind narrow adapters. Stage logic should depend on Narratio-level interfaces and data structures, not on external SDK types, subprocess argument construction, or transport-specific details.
|
|
||||||
|
|
||||||
### Standard library preference
|
|
||||||
|
|
||||||
Prefer the Go standard library. Add dependencies only when they provide substantial value, are necessary for an external integration, or are a widely used de facto standard.
|
|
||||||
|
|
||||||
Accepted examples include a YAML library for configuration and the AWS SDK for S3-compatible storage.
|
|
||||||
|
|
||||||
### Explicit orchestration
|
|
||||||
|
|
||||||
The pipeline should remain stage-driven and explicit. New behavior should be added through clear stage, adapter, config, or manifest contracts rather than implicit side effects or generic workflow abstraction.
|
|
||||||
|
|
||||||
## Stage Design
|
|
||||||
|
|
||||||
Each stage should have a clear scope of responsibility.
|
|
||||||
|
|
||||||
A stage should define:
|
|
||||||
|
|
||||||
- its purpose;
|
|
||||||
- required input state;
|
|
||||||
- produced output state;
|
|
||||||
- config fields it consumes;
|
|
||||||
- external adapters it uses;
|
|
||||||
- manifest refs it reads or writes;
|
|
||||||
- skip, force, and resume behavior;
|
|
||||||
- failure behavior;
|
|
||||||
- tests that protect its contract.
|
|
||||||
|
|
||||||
Stages should avoid reaching across boundaries. If shared behavior is needed, prefer a helper or service with a narrow interface over duplicating ad hoc logic between stages.
|
|
||||||
|
|
||||||
## Transactionality and Resume
|
|
||||||
|
|
||||||
A stage should behave transactionally.
|
|
||||||
|
|
||||||
A stage is complete only when its outputs have been written, validated, and recorded in the manifest. If a stage fails, Narratio should preserve enough local state for inspection, recovery, and resume.
|
|
||||||
|
|
||||||
A failed or incomplete run must not be treated as successful. Later stages should depend on manifest-recorded success, not merely on incidental files existing on disk.
|
|
||||||
|
|
||||||
## Manifest Model
|
|
||||||
|
|
||||||
The manifest is the durable local ledger for a run.
|
|
||||||
|
|
||||||
It should record:
|
|
||||||
|
|
||||||
- run identity;
|
|
||||||
- stage status;
|
|
||||||
- input and output refs;
|
|
||||||
- logs and generated config refs;
|
|
||||||
- checksums or provenance where useful;
|
|
||||||
- non-secret adapter and archive metadata.
|
|
||||||
|
|
||||||
Resume behavior should be manifest-driven. Filesystem state may be inspected and validated, but it should not replace manifest stage state as the source of run progress.
|
|
||||||
|
|
||||||
## Adapter Boundaries
|
|
||||||
|
|
||||||
Adapters own external integration details.
|
|
||||||
|
|
||||||
Expected boundaries:
|
|
||||||
|
|
||||||
- WhisperX HTTP details stay in the WhisperX adapter.
|
|
||||||
- Seriatim CLI construction stays in the Seriatim adapter.
|
|
||||||
- Audita CLI construction stays in the Audita adapter.
|
|
||||||
- Scriptorium CLI construction stays in the Scriptorium adapter.
|
|
||||||
- Object-storage details stay behind the storage adapter interface.
|
|
||||||
- AWS SDK types stay inside the S3 storage implementation.
|
|
||||||
|
|
||||||
Stage code should express intent in Narratio terms and call adapters through narrow contracts.
|
|
||||||
|
|
||||||
## Configuration Philosophy
|
|
||||||
|
|
||||||
Configuration should be strict, explicit, and operator-friendly.
|
|
||||||
|
|
||||||
Principles:
|
|
||||||
|
|
||||||
- YAML decoding should reject unknown fields.
|
|
||||||
- Defaults should be centralized and testable.
|
|
||||||
- Empty configured values should not silently override meaningful defaults.
|
|
||||||
- Session templating should remain narrow and deterministic.
|
|
||||||
- Template support should serve operator convenience, not become a general configuration language.
|
|
||||||
|
|
||||||
Narratio should not become a secondary configuration system for downstream tools. Seriatim, Audita, and Scriptorium should own their runtime defaults wherever practical. Narratio should pass required stage-contract paths and explicit operator overrides.
|
|
||||||
|
|
||||||
## Path and Storage Discipline
|
|
||||||
|
|
||||||
Local and remote paths are part of Narratio’s application contract.
|
|
||||||
|
|
||||||
Code should use centralized path helpers for workspace, spool, session, run, artifact, log, config, and archive paths. Stages should avoid reconstructing canonical paths through scattered string concatenation.
|
|
||||||
|
|
||||||
Storage backends should receive explicit bucket-relative keys. Storage implementations should not infer campaign, session, run, or root-prefix semantics.
|
|
||||||
|
|
||||||
## Archive Invariants
|
|
||||||
|
|
||||||
Archive behavior must preserve a clear commit boundary.
|
|
||||||
|
|
||||||
A remote run is current only after the archive stage has successfully uploaded the run record, required promoted outputs, `current/manifest.json`, and finally `current/run_id.txt`.
|
|
||||||
|
|
||||||
`current/run_id.txt` is the final remote commit marker and must be written last.
|
|
||||||
|
|
||||||
Failed, incomplete, skipped, or uncommitted archive attempts must not be presented as current remote state. Local cleanup is permitted only after successful archive commit and only when explicitly configured.
|
|
||||||
|
|
||||||
## Security and Privacy
|
|
||||||
|
|
||||||
Narratio handles private campaign material.
|
|
||||||
|
|
||||||
Rules:
|
|
||||||
|
|
||||||
- Do not store raw secrets in pipeline or session YAML.
|
|
||||||
- Use environment variable names or secret-file references for secret handling.
|
|
||||||
- Do not write raw secret values to manifests, logs, generated configs, or archive metadata.
|
|
||||||
- Treat transcripts, generated artifacts, prompts, reports, and logs as potentially sensitive.
|
|
||||||
- Avoid logging transcript or prompt content unless there is a deliberate diagnostic reason.
|
|
||||||
|
|
||||||
## Diagnostics
|
|
||||||
|
|
||||||
Diagnostics should be durable and discoverable, but distinct from canonical outputs.
|
|
||||||
|
|
||||||
Logs, reports, generated invocation/config files, and render-debug files support debugging. Transcript tiers and configured artifacts are pipeline products.
|
|
||||||
|
|
||||||
Manifest refs should preserve that distinction.
|
|
||||||
|
|
||||||
## Determinism
|
|
||||||
|
|
||||||
Where practical, Narratio should prefer deterministic behavior:
|
|
||||||
|
|
||||||
- stable local path layout;
|
|
||||||
- stable remote key layout;
|
|
||||||
- sorted upload order;
|
|
||||||
- predictable generated config files;
|
|
||||||
- repeatable command construction;
|
|
||||||
- tests that do not depend on live external services.
|
|
||||||
|
|
||||||
Run IDs and timestamps may be intentionally variable, but surrounding behavior should remain testable.
|
|
||||||
|
|
||||||
## Testing Expectations
|
|
||||||
|
|
||||||
Core behavior should be testable without live external services.
|
|
||||||
|
|
||||||
Tests should cover:
|
|
||||||
|
|
||||||
- config loading, defaults, and validation;
|
|
||||||
- CLI parsing and command construction;
|
|
||||||
- path helpers;
|
|
||||||
- manifest transitions;
|
|
||||||
- stage success, failure, skip, and resume behavior;
|
|
||||||
- adapter command construction;
|
|
||||||
- fake storage behavior;
|
|
||||||
- archive commit ordering;
|
|
||||||
- example config validity where practical.
|
|
||||||
|
|
||||||
Live S3, WhisperX, LLM, or subprocess integration tests should be explicit integration tests, not required for ordinary unit test runs.
|
|
||||||
|
|
||||||
## Documentation Expectations
|
|
||||||
|
|
||||||
Documentation must follow `docs/documentation/policy.md`.
|
|
||||||
|
|
||||||
Current behavior belongs in user-facing docs and `docs/internal/`. Future, planned, aspirational, experimental, or unimplemented work belongs only under `docs/roadmap/`.
|
|
||||||
|
|
||||||
`docs/architecture.md` should remain concise and principle-focused. It should not duplicate the full config reference, CLI reference, operations guide, or internal stage documentation.
|
|
||||||
|
|
||||||
## Non-Goals
|
|
||||||
|
|
||||||
Narratio is not:
|
|
||||||
|
|
||||||
- a generic DAG or workflow engine;
|
|
||||||
- a replacement configuration layer for Seriatim, Audita, or Scriptorium;
|
|
||||||
- a storage backend abstraction beyond the needs of this pipeline;
|
|
||||||
- a place to embed raw secrets;
|
|
||||||
- a place for stage logic to depend directly on AWS SDK types or downstream tool internals;
|
|
||||||
- a prompt-authoring system.
|
|
||||||
491
docs/cli.md
491
docs/cli.md
@@ -1,60 +1,202 @@
|
|||||||
# CLI
|
# CLI Reference
|
||||||
|
|
||||||
## Shortest Useful Command
|
## Shortest Useful Command
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio run --session-id 2026-04-04
|
narratio run 2026-04-04
|
||||||
```
|
```
|
||||||
|
|
||||||
This command uses default config discovery for `pipeline.yml` and `session.yml`; both files must be discoverable unless you pass explicit `--config` and `--session` paths.
|
This runs the canonical full pipeline for session `2026-04-04`.
|
||||||
|
|
||||||
## Command Overview
|
## Command Overview
|
||||||
|
|
||||||
Implemented commands:
|
Top-level commands:
|
||||||
|
|
||||||
- `run`: execute pipeline stages and persist manifest state.
|
- `version`: print the Narratio build version.
|
||||||
- `plan`: validate config, prepare workspace layout, and print stage run/skip decisions.
|
- `run <session_id>`: run all or one contiguous range of the canonical stage order.
|
||||||
- `resume`: continue from first non-succeeded stage unless forced.
|
- `regenerate-artifacts <session_id>`: force-run extraction through analysis.
|
||||||
- `status`: read and print stage statuses from an existing manifest.
|
- `run-stage <stage> <session_id>`: run one stage.
|
||||||
- `run-stage`: execute exactly one stage.
|
- `analyze <session_id>`: force-run analyze.
|
||||||
|
- `publish <session_id>`: force-run publish.
|
||||||
|
- `clean <session_id>` or `clean --all`: remove local work/spool state.
|
||||||
|
- `session <subcommand>`: session helper commands.
|
||||||
|
- `config <subcommand>`: validate, display, source-trace, or compare resolved pipeline configuration.
|
||||||
|
|
||||||
Unknown commands print usage and exit non-zero.
|
Session subcommands:
|
||||||
|
|
||||||
For config semantics, see [docs/config.md](./config.md). For operator lifecycle and recovery, see [docs/operations.md](./operations.md).
|
- `session init <session_id>`
|
||||||
|
- `session plan <session_id>`
|
||||||
|
- `session validate <session_id>`
|
||||||
|
- `session status <session_id>`
|
||||||
|
- `session restore <session_id>`
|
||||||
|
- `session artifacts <session_id>`
|
||||||
|
- `session locks <session_id>`
|
||||||
|
- `session locks add <session_id> <source>`
|
||||||
|
- `session locks remove <session_id> <source>`
|
||||||
|
|
||||||
## Complete Flag Reference
|
## Common Config Flags
|
||||||
|
|
||||||
|
Most session-aware commands accept:
|
||||||
|
|
||||||
|
- `--config <pipeline.yml>`
|
||||||
|
- `--campaign <id>`
|
||||||
|
- `--campaign-file <campaign.yml>`
|
||||||
|
- `--session <session.yml>`
|
||||||
|
- `--session-id <session_id>`
|
||||||
|
- `--previous-session-id <session_id>`
|
||||||
|
- `--profile <name>`
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
|
||||||
|
- `--campaign` and `--campaign-file` are mutually exclusive.
|
||||||
|
- `--session` is not used by `session init`.
|
||||||
|
- if both positional `<session_id>` and `--session-id` are provided, values must match.
|
||||||
|
- `--previous-session-id` is a strict expectation: the selected session file
|
||||||
|
must contain the same `previous_session_id`.
|
||||||
|
- `--profile` selects a declared pipeline profile. It may be supplied once;
|
||||||
|
an explicit empty or unknown value fails configuration resolution. When it is
|
||||||
|
omitted, a declared `default_profile` is used. The same selection applies to
|
||||||
|
all common-flag commands, including `regenerate-artifacts`.
|
||||||
|
- `clean --all` cannot be combined with campaign/session selectors.
|
||||||
|
- notification delivery is currently limited to the configured `noop` mode; see
|
||||||
|
the [configuration reference](./config.md#notifications).
|
||||||
|
|
||||||
|
## Session ID Input Rules
|
||||||
|
|
||||||
|
Session-aware commands accept one of these forms:
|
||||||
|
|
||||||
|
- positional session ID: `... <session_id>`
|
||||||
|
- compatibility flag: `... --session-id <session_id>`
|
||||||
|
|
||||||
|
When both are present, command parsing requires an exact match.
|
||||||
|
|
||||||
|
Commands with additional positionals keep their command-specific order:
|
||||||
|
|
||||||
|
- `run-stage <stage> <session_id>` or `run-stage <stage> --session-id <session_id>`
|
||||||
|
- `session locks add <session_id> <source>` or `session locks add --session-id <session_id> <source>`
|
||||||
|
- `session locks remove <session_id> <source>` or `session locks remove --session-id <session_id> <source>`
|
||||||
|
|
||||||
|
## Command Reference
|
||||||
|
|
||||||
|
### `config validate`, `config show`, `config sources`, and `config diff`
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio config validate [--config <pipeline.yml>] [--campaign <id> | --campaign-file <campaign.yml>] [--profile <name>]
|
||||||
|
narratio config show [--config <pipeline.yml>] [--campaign <id> | --campaign-file <campaign.yml>] [--profile <name>]
|
||||||
|
narratio config sources [--config <pipeline.yml>] [--campaign <id> | --campaign-file <campaign.yml>] [--profile <name>]
|
||||||
|
narratio config diff <left-profile> <right-profile> [--config <pipeline.yml>] [--campaign <id> | --campaign-file <campaign.yml>]
|
||||||
|
```
|
||||||
|
|
||||||
|
These commands resolve the selected profile, defaults, ordinary paths, and—if
|
||||||
|
a campaign is selected—the campaign-owned party. They neither discover or load
|
||||||
|
a session nor create a workspace, manifest, run, lock, adapter, remote
|
||||||
|
connection, or credential environment.
|
||||||
|
|
||||||
|
Campaign selection is optional for a pipeline without party-driven artifact
|
||||||
|
families. A pipeline with `scriptorium.artifact_families` needs a selected or
|
||||||
|
configured default campaign so Narratio can expand its concrete artifacts and
|
||||||
|
publish rules. `--campaign` and `--campaign-file` remain mutually exclusive.
|
||||||
|
Session, range, force, and artifact-execution flags are not accepted.
|
||||||
|
|
||||||
|
`config validate` writes a concise root-path, selected-profile (or `none`), and
|
||||||
|
effective-digest summary after successful complete validation. `config show`
|
||||||
|
writes one deterministic, secret-free YAML document containing defaulted and
|
||||||
|
expanded concrete configuration. It omits composition declarations, artifact
|
||||||
|
family declarations, and runtime provenance.
|
||||||
|
|
||||||
|
`config sources` reports the same fully validated resolution without printing
|
||||||
|
effective values. Its header identifies the root, ordered imports, selected
|
||||||
|
profile and overlay, selected campaign, party mode/source, and digest. The
|
||||||
|
remaining tab-separated records are sorted as `path`, `role`, and `source`.
|
||||||
|
Roles distinguish root, import, profile, centralized default, campaign, party,
|
||||||
|
legacy-player, and generated family ownership. A generated party member has
|
||||||
|
one family record and one party record at the same logical path. The output
|
||||||
|
never reads or prints secret values.
|
||||||
|
|
||||||
|
`config diff` resolves both supplied profile names from one parsed root source
|
||||||
|
set and compares their fully resolved, secret-free effective mappings. It does
|
||||||
|
not accept `--profile`; the two positional names must be distinct, declared
|
||||||
|
profiles. When party-driven families are present, both profiles must resolve to
|
||||||
|
the same selected campaign and party. Use `--campaign-file` if profile-specific
|
||||||
|
campaign configuration would otherwise select different files.
|
||||||
|
|
||||||
|
Equal profiles print `no differences`. Otherwise, sorted tab-separated records
|
||||||
|
use one of these forms, with compact JSON values:
|
||||||
|
|
||||||
|
```text
|
||||||
|
added <path> <right-value>
|
||||||
|
removed <path> <left-value>
|
||||||
|
changed <path> <left-value> <right-value>
|
||||||
|
```
|
||||||
|
|
||||||
|
Mappings are flattened to their logical field paths; lists remain one atomic
|
||||||
|
value. The command compares defaulted concrete artifacts and publish rules, not
|
||||||
|
profile names, source-file layout, or formatting. It succeeds when differences
|
||||||
|
are found, making it suitable for review and migration checks.
|
||||||
|
|
||||||
|
### `version`
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio version
|
||||||
|
```
|
||||||
|
|
||||||
|
Official release binaries report their exact Git tag. Binaries built directly
|
||||||
|
from source without release linker metadata report `dev`.
|
||||||
|
|
||||||
### `run`
|
### `run`
|
||||||
|
|
||||||
- `--config <path>`: optional explicit `pipeline.yml` path.
|
```bash
|
||||||
- `--session <path>`: optional explicit `session.yml` path.
|
narratio run <session_id> [--from <stage>] [--through <stage>] [--force] [--artifacts <name[,name...]>] [...common config flags]
|
||||||
- `--session-id <value>`: session template variable value.
|
```
|
||||||
- `--force`: force stage execution.
|
|
||||||
- `--artifacts <names>`: analyze artifact keys to execute (repeatable or comma-separated).
|
|
||||||
|
|
||||||
### `plan`
|
Behavior:
|
||||||
|
|
||||||
- `--config <path>`
|
- evaluates one inclusive contiguous range of the canonical stage order;
|
||||||
- `--session <path>`
|
- defaults an omitted `--from` to `prepare` and an omitted `--through` to
|
||||||
- `--session-id <value>`
|
`notify`, so omitting both retains full-pipeline behavior;
|
||||||
- `--force`
|
- rejects unknown endpoints and a `--from` endpoint after `--through`;
|
||||||
|
- runs `render` before `extract`; an omitted or disabled Notarius
|
||||||
|
configuration records an explicit `notarius_disabled` self-skip;
|
||||||
|
- skips already-succeeded stages unless `--force` is set or a stage-specific
|
||||||
|
resume check finds its durable result obsolete;
|
||||||
|
- applies `--force` only to stages in the selected range;
|
||||||
|
- rejects repeated `--from`, `--through`, or `--force` options, including
|
||||||
|
`--name=value` spellings;
|
||||||
|
- continues interrupted or partially completed sessions by running non-succeeded stages;
|
||||||
|
- writes session and run manifests.
|
||||||
|
- reports the resolved profile (or `none`) and effective configuration digest.
|
||||||
|
|
||||||
### `resume`
|
When `--artifacts` is present, the selected range must contain `analyze` or
|
||||||
|
`publish`. Either consumer is sufficient, including a one-stage range.
|
||||||
|
|
||||||
- `--config <path>`
|
### `regenerate-artifacts`
|
||||||
- `--session <path>`
|
|
||||||
- `--session-id <value>`
|
```bash
|
||||||
- `--force`
|
narratio regenerate-artifacts <session_id> [--artifacts <name[,name...]>] [...common config flags]
|
||||||
- `--artifacts <names>`: analyze artifact keys to execute (repeatable or comma-separated).
|
```
|
||||||
|
|
||||||
|
Exactly equivalent to:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run <session_id> --force --from extract --through analyze [caller options]
|
||||||
|
```
|
||||||
|
|
||||||
|
The command always reruns extraction. Analysis rebuilds the selected configured
|
||||||
|
artifacts and any prerequisites required by those targets; without
|
||||||
|
`--artifacts`, it uses the normal default analysis selection. Publish and notify
|
||||||
|
never run. Common session/configuration options and repeatable artifact values
|
||||||
|
pass through unchanged.
|
||||||
|
|
||||||
|
Because the expansion owns `--force`, `--from`, and `--through`, callers cannot
|
||||||
|
supply those options. The shared `run` parser reports them as duplicate
|
||||||
|
singleton flags. The alias has no private execution options or behavior, and
|
||||||
|
runtime diagnostics may identify the operation as `run`.
|
||||||
|
|
||||||
### `run-stage`
|
### `run-stage`
|
||||||
|
|
||||||
- `--config <path>`
|
```bash
|
||||||
- `--session <path>`
|
narratio run-stage <stage> <session_id> [--force] [--artifacts <name[,name...]>] [...common config flags]
|
||||||
- `--session-id <value>`
|
```
|
||||||
- `--force`
|
|
||||||
- `--artifacts <names>`: analyze artifact keys to execute (repeatable or comma-separated).
|
|
||||||
- positional `<stage>`: required stage name.
|
|
||||||
|
|
||||||
Valid stage names:
|
Valid stage names:
|
||||||
|
|
||||||
@@ -64,159 +206,232 @@ Valid stage names:
|
|||||||
- `polish`
|
- `polish`
|
||||||
- `normalize`
|
- `normalize`
|
||||||
- `trim`
|
- `trim`
|
||||||
|
- `render`
|
||||||
|
- `extract`
|
||||||
- `analyze`
|
- `analyze`
|
||||||
- `archive`
|
- `publish`
|
||||||
- `notify`
|
- `notify`
|
||||||
|
|
||||||
### `status`
|
Rules:
|
||||||
|
|
||||||
- `--manifest <path>`: required manifest path.
|
- `--artifacts` is accepted only for `analyze` and `publish` stage targets.
|
||||||
|
|
||||||
## Command Reference
|
### `analyze`
|
||||||
|
|
||||||
### `run`
|
|
||||||
|
|
||||||
Purpose:
|
|
||||||
- Execute configured stages in canonical order.
|
|
||||||
|
|
||||||
Syntax:
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio run [--config <pipeline.yml>] [--session <session.yml>] [--session-id <id>] [--force] [--artifacts <name[,name...]>]
|
narratio analyze <session_id> [--artifacts <name[,name...]>] [...common config flags]
|
||||||
```
|
```
|
||||||
|
|
||||||
Success output:
|
Equivalent to:
|
||||||
- `narratio run: session <session_id>; executed=<n> skipped=<n>; manifest=<path>`
|
|
||||||
|
|
||||||
Common failure cases:
|
|
||||||
- missing default config/session paths when flags omitted.
|
|
||||||
- invalid template/rendered session mismatch.
|
|
||||||
- unknown/invalid `--artifacts` value.
|
|
||||||
- `--artifacts` with unknown configured artifact key.
|
|
||||||
|
|
||||||
### `plan`
|
|
||||||
|
|
||||||
Purpose:
|
|
||||||
- Validate config, load secrets (if configured), prepare workdir, and print stage run/skip decisions.
|
|
||||||
|
|
||||||
Syntax:
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio plan [--config <pipeline.yml>] [--session <session.yml>] [--session-id <id>] [--force]
|
narratio run-stage analyze <session_id> --force [...common config flags]
|
||||||
```
|
```
|
||||||
|
|
||||||
Success output includes:
|
### `publish`
|
||||||
- `narratio plan: workdir prepared at <path>`
|
|
||||||
- one line per stage (`<stage>: run|skip`)
|
|
||||||
- `totals: run=<n> skip=<n>`
|
|
||||||
|
|
||||||
Common failure cases:
|
|
||||||
- same config/session discovery and validation failures as `run`.
|
|
||||||
- secrets directory read failures when `pipeline.secrets.env_dir` is configured.
|
|
||||||
|
|
||||||
### `resume`
|
|
||||||
|
|
||||||
Purpose:
|
|
||||||
- Continue from session-manifest stage status.
|
|
||||||
|
|
||||||
Syntax:
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio resume [--config <pipeline.yml>] [--session <session.yml>] [--session-id <id>] [--force] [--artifacts <name[,name...]>]
|
narratio publish <session_id> [--artifacts <name[,name...]>] [...common config flags]
|
||||||
```
|
```
|
||||||
|
|
||||||
Success output:
|
Equivalent to:
|
||||||
- `narratio resume: session <session_id> has no remaining stages`
|
|
||||||
- or `narratio resume: session <session_id>; executed=<n> skipped=<n>; manifest=<path>`
|
|
||||||
|
|
||||||
Common failure cases:
|
|
||||||
- same discovery/template/validation failures as `run`.
|
|
||||||
- manifest load errors when existing manifest is unreadable.
|
|
||||||
- invalid or unknown artifact selections.
|
|
||||||
|
|
||||||
### `status`
|
|
||||||
|
|
||||||
Purpose:
|
|
||||||
- Inspect one manifest file without executing stages.
|
|
||||||
|
|
||||||
Syntax:
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio status --manifest <manifest.json>
|
narratio run-stage publish <session_id> --force [...common config flags]
|
||||||
```
|
```
|
||||||
|
|
||||||
Success output includes:
|
### `clean`
|
||||||
- `session_id: <id>`
|
|
||||||
- `updated_at: <timestamp>`
|
|
||||||
- `stages:` entries (`- <stage>: <status>`)
|
|
||||||
|
|
||||||
Common failure cases:
|
|
||||||
- missing `--manifest`.
|
|
||||||
- unreadable or invalid manifest path.
|
|
||||||
|
|
||||||
### `run-stage`
|
|
||||||
|
|
||||||
Purpose:
|
|
||||||
- Execute exactly one stage.
|
|
||||||
|
|
||||||
Syntax:
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio run-stage [--config <pipeline.yml>] [--session <session.yml>] [--session-id <id>] [--force] [--artifacts <name[,name...]>] <stage>
|
narratio clean <session_id> [--dry-run] [--clear-cache] [...common config flags]
|
||||||
|
narratio clean --all [--dry-run] [--clear-cache] [--config <pipeline.yml>]
|
||||||
```
|
```
|
||||||
|
|
||||||
Success output:
|
Behavior:
|
||||||
- `narratio run-stage: stage=<name> executed=<n> skipped=<n> force=<true|false>; manifest=<path>`
|
|
||||||
|
|
||||||
`--artifacts` behavior:
|
- session mode removes the selected session's local work and spool state;
|
||||||
- accepted only when `<stage>` is `analyze`.
|
- `--all` removes all local session work and spool state;
|
||||||
- names are normalized (trimmed, deduplicated, sorted).
|
- cache remains unless `--clear-cache` is provided.
|
||||||
- unknown configured artifact keys fail.
|
|
||||||
|
|
||||||
Common failure cases:
|
See [Operations: Cleanup](./operations.md#cleanup) for deletion scope and
|
||||||
- missing stage positional arg.
|
post-publish cleanup behavior.
|
||||||
- unknown stage name.
|
|
||||||
- using `--artifacts` with any non-`analyze` stage.
|
### `session plan`
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session plan <session_id> [--from <stage>] [--through <stage>] [--force] [--artifacts <name[,name...]>] [...common config flags]
|
||||||
|
```
|
||||||
|
|
||||||
|
Uses the same inclusive bounds, endpoint validation, force scope, and artifact
|
||||||
|
selection contract as `run`. It validates config and prints run/skip decisions
|
||||||
|
for selected stages only without creating the local workdir or changing the
|
||||||
|
manifest. Resume-capable selected stages are checked against durable evidence.
|
||||||
|
The output includes the resolved profile (or `none`) and effective configuration
|
||||||
|
digest without writing provenance or any manifest state.
|
||||||
|
For `analyze`, the preview also lists explicit targets, prerequisite-only work,
|
||||||
|
execution order, and reusable current artifacts with concise reasons. These
|
||||||
|
artifact decisions come from the same reconciliation and work planner used by
|
||||||
|
execution; the preview does not predict output identities.
|
||||||
|
|
||||||
|
### `session validate`
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session validate <session_id> [...common config flags]
|
||||||
|
```
|
||||||
|
|
||||||
|
Read-only preflight checks for config validity, required inputs, audio mode, previous-session requirements, publish outputs, and effective locks.
|
||||||
|
|
||||||
|
### `session status`
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session status <session_id> [...common config flags]
|
||||||
|
```
|
||||||
|
|
||||||
|
Prints local manifest state and, when storage is available, status for the
|
||||||
|
pointer-selected remote commit and its declared published outputs.
|
||||||
|
|
||||||
|
### `session init`
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session init <session_id> --output ./session.yml [options]
|
||||||
|
narratio session init <session_id> --remote [options]
|
||||||
|
```
|
||||||
|
|
||||||
|
Required target selection:
|
||||||
|
|
||||||
|
- exactly one of:
|
||||||
|
- `--output <path>`
|
||||||
|
- `--remote`
|
||||||
|
|
||||||
|
Options:
|
||||||
|
|
||||||
|
- `--config <pipeline.yml>`
|
||||||
|
- `--campaign <id>` or `--campaign-file <campaign.yml>`
|
||||||
|
- `--previous-session-id <id>`
|
||||||
|
- `--date <YYYY-MM-DD>`
|
||||||
|
- `--title <text>`
|
||||||
|
- `--audio-dir <path>`
|
||||||
|
- `--audio-s3-prefix <prefix>`
|
||||||
|
- `--force`
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
|
||||||
|
- `--audio-dir` and `--audio-s3-prefix` are mutually exclusive.
|
||||||
|
- if campaign `session_template_file` is configured, `session init` renders it.
|
||||||
|
- generated session YAML must be concrete (no unresolved `{{ ... }}` placeholders).
|
||||||
|
|
||||||
|
### `session restore`
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session restore <session_id> [--dry-run] [--force] [--include-audio] [...common config flags]
|
||||||
|
```
|
||||||
|
|
||||||
|
Behavior:
|
||||||
|
|
||||||
|
- discovers committed remote current state;
|
||||||
|
- plans local restores;
|
||||||
|
- writes an execution report;
|
||||||
|
- blocks unresolved conflicts. `--force` permits replacement only of eligible
|
||||||
|
regular files.
|
||||||
|
|
||||||
|
See [Operations: Restore Workflow](./operations.md#restore-workflow) for the
|
||||||
|
default restore scope, report location, and conflict-handling workflow.
|
||||||
|
|
||||||
|
### `session artifacts`
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session artifacts <session_id> [--remote] [...common config flags]
|
||||||
|
```
|
||||||
|
|
||||||
|
Lists effective built-in, configured Scriptorium, and configured extraction
|
||||||
|
sources; reports planned, available, unavailable, and published state without
|
||||||
|
reading payload bodies; and includes publish rules, lock state, and optional
|
||||||
|
remote published-state availability.
|
||||||
|
|
||||||
|
### `session locks`
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session locks <session_id> [...common config flags]
|
||||||
|
narratio session locks add <session_id> <source> [--reason <text>] [--force] [...common config flags]
|
||||||
|
narratio session locks remove <session_id> <source> [...common config flags]
|
||||||
|
```
|
||||||
|
|
||||||
|
Behavior:
|
||||||
|
|
||||||
|
- list mode reports the effective merge of static and remote locks;
|
||||||
|
- add/remove mutate only remote locks;
|
||||||
|
- static locks from pipeline config cannot be removed by CLI commands.
|
||||||
|
|
||||||
|
See [Operations: Publish Locks](./operations.md#publish-locks) for lock storage
|
||||||
|
and precedence.
|
||||||
|
|
||||||
|
## `--artifacts` Selection Rules
|
||||||
|
|
||||||
|
An artifact-family key selects all of its concrete character members. A
|
||||||
|
concrete generated key selects only that member; mixed family and concrete
|
||||||
|
selection is deduplicated and executed as concrete keys. The resulting plan
|
||||||
|
and command output identify both the concrete key and, where applicable, its
|
||||||
|
family and character ID.
|
||||||
|
|
||||||
|
- accepted on `run`, `session plan`, `run-stage`, `analyze`, and `publish`;
|
||||||
|
- repeatable and comma-separated values are combined, surrounding whitespace
|
||||||
|
is removed, and duplicate names are collapsed;
|
||||||
|
- names must exist in `pipeline.scriptorium.artifacts`;
|
||||||
|
- empty entries are invalid;
|
||||||
|
- on `run-stage`, only `analyze` and `publish` accept the option.
|
||||||
|
|
||||||
|
Effects:
|
||||||
|
|
||||||
|
- selects explicit analyze targets; required configured prerequisites may be
|
||||||
|
reused or rebuilt before them;
|
||||||
|
- filters publish rules that source `narratio.artifact.<name>`;
|
||||||
|
- does not filter built-in transcript/bounds or explicitly configured
|
||||||
|
`narratio.extraction.<name>` publish sources; and
|
||||||
|
- does not select or filter Notarius lanes.
|
||||||
|
|
||||||
## Common Workflows
|
## Common Workflows
|
||||||
|
|
||||||
Default-discovery run:
|
Run full pipeline:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio run --session-id 2026-04-04
|
narratio run 2026-04-04
|
||||||
```
|
```
|
||||||
|
|
||||||
Run only selected analyze artifacts:
|
Dry-run restore plan:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio run --session-id 2026-04-04 --artifacts session_recap,player_handout
|
narratio session restore 2026-04-04 --dry-run
|
||||||
```
|
```
|
||||||
|
|
||||||
Resume with selected analyze artifacts:
|
Generate a concrete session file from template/default structure:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio resume --session-id 2026-04-04 --artifacts player_handout
|
narratio session init 2026-04-04 --output ./session.yml --date 2026-04-04 --title "Session 12"
|
||||||
```
|
```
|
||||||
|
|
||||||
Run only analyze stage with selected artifacts:
|
Force publish only:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio run-stage --session-id 2026-04-04 --artifacts player_handout analyze
|
narratio publish 2026-04-04
|
||||||
```
|
```
|
||||||
|
|
||||||
## Diagnostic / Recovery Commands
|
Regenerate post-transcript artifacts without publishing:
|
||||||
|
|
||||||
Inspect stage status:
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio status --manifest <manifest.json>
|
narratio regenerate-artifacts 2026-04-04 --artifacts session_recap,player_handout
|
||||||
```
|
```
|
||||||
|
|
||||||
Get manifest path from previous output:
|
## Output And Exit Behavior
|
||||||
- `run`, `resume`, and `run-stage` print `manifest=<path>` on success.
|
|
||||||
|
|
||||||
## `--artifacts` and `--force`
|
- Successful commands write their result or summary to standard output and
|
||||||
|
exit with status `0`.
|
||||||
|
- Command failures and invalid invocations write an error to standard error and
|
||||||
|
exit with status `1`.
|
||||||
|
- An unknown top-level command also prints the top-level usage summary to
|
||||||
|
standard error.
|
||||||
|
- `session restore --help` prints its command-specific usage and exits with
|
||||||
|
status `0`.
|
||||||
|
|
||||||
- `--artifacts` filters which configured artifacts are executable when analyze runs.
|
Output is intended for operator inspection. Narratio does not currently offer
|
||||||
- `--artifacts` does not imply `--force`.
|
a machine-readable CLI output mode; durable machine-readable state is recorded
|
||||||
- If analyze is already `succeeded` and `--force` is not set, runner-level skip still applies.
|
in manifests and reports described in [Operations](./operations.md).
|
||||||
|
|||||||
670
docs/config.md
670
docs/config.md
@@ -1,149 +1,295 @@
|
|||||||
# Configuration
|
# Configuration Reference
|
||||||
|
|
||||||
## 1. Overview
|
## Purpose
|
||||||
|
|
||||||
Narratio loads two YAML files:
|
Narratio resolves three YAML documents:
|
||||||
|
|
||||||
- `pipeline.yml`: pipeline-level runtime configuration.
|
- `pipeline.yml`: pipeline/runtime settings
|
||||||
- `session.yml`: per-session metadata and input selection.
|
- `campaign.yml`: campaign identity and stable input defaults
|
||||||
|
- `session.yml`: session identity, metadata, and audio source selection
|
||||||
|
|
||||||
These commands load and validate both files before running:
|
## Discovery and Selection
|
||||||
|
|
||||||
- `narratio run`
|
### `pipeline.yml`
|
||||||
- `narratio plan`
|
|
||||||
- `narratio resume`
|
|
||||||
- `narratio run-stage`
|
|
||||||
|
|
||||||
Behavior:
|
When `--config` is omitted, search order is:
|
||||||
|
|
||||||
- strict YAML decode is enabled (`KnownFields(true)`): unknown fields fail.
|
1. `/usr/local/etc/narratio/pipeline.yml`
|
||||||
- session templates render before session YAML decode.
|
2. `/etc/narratio/pipeline.yml`
|
||||||
- defaults are applied for optional pipeline fields.
|
|
||||||
- validation enforces required fields, value formats, and cross-field constraints.
|
|
||||||
|
|
||||||
## 2. Config file discovery
|
### `campaign.yml`
|
||||||
|
|
||||||
Pipeline config lookup for `run`, `plan`, `resume`, and `run-stage`:
|
Selection rules:
|
||||||
|
|
||||||
- If `--config <path>` is provided, that path is used.
|
- if `--campaign-file` is set, use that path;
|
||||||
- If omitted, Narratio searches in order:
|
- else if `--campaign <id>` is set, use `{pipeline.campaigns.root}/{id}/campaign.yml`;
|
||||||
1. `/usr/local/etc/narratio/pipeline.yml`
|
- else use `{pipeline.campaigns.root}/{pipeline.campaigns.default_campaign_id}/campaign.yml`.
|
||||||
2. `/etc/narratio/pipeline.yml`
|
|
||||||
- First existing file wins.
|
|
||||||
|
|
||||||
## 3. Session file discovery and templating
|
### `session.yml`
|
||||||
|
|
||||||
Session config lookup for `run`, `plan`, `resume`, and `run-stage`:
|
When `--session` is omitted, local search order is:
|
||||||
|
|
||||||
- If `--session <path>` is provided, that path is used.
|
1. `/usr/local/etc/narratio/session.yml`
|
||||||
- If omitted, Narratio searches in order:
|
2. `/etc/narratio/session.yml`
|
||||||
1. `./session.yml`
|
|
||||||
2. `/usr/local/etc/narratio/session.yml`
|
|
||||||
3. `/etc/narratio/session.yml`
|
|
||||||
- First existing file wins.
|
|
||||||
|
|
||||||
Template behavior:
|
If local session discovery fails and a `session_id` is known, Narratio attempts remote session loading from:
|
||||||
|
|
||||||
- Supported placeholders:
|
- `{root_prefix}/campaigns/{campaign}/sessions/{session_id}/session.yml`
|
||||||
- `{{session_id}}`
|
|
||||||
- `{{ session_id }}`
|
|
||||||
- `--session-id <value>` supplies the placeholder value.
|
|
||||||
- unresolved placeholders fail load.
|
|
||||||
- if rendered `session_id` mismatches `--session-id`, load fails.
|
|
||||||
|
|
||||||
## 4. Minimal pipeline config
|
using configured object storage.
|
||||||
|
|
||||||
|
The downloaded remote session file is command-scoped: Narratio removes it after
|
||||||
|
the command finishes and records only the remote object provenance alongside
|
||||||
|
the durable copied session input.
|
||||||
|
|
||||||
|
### Read-only effective pipeline inspection
|
||||||
|
|
||||||
|
`narratio config validate`, `narratio config show`, and `narratio config
|
||||||
|
sources` use the same `--config`, `--campaign`, `--campaign-file`, and
|
||||||
|
`--profile` selection rules as pipeline commands, but do not select, discover,
|
||||||
|
or load a session. They do not read credential values or create runtime state.
|
||||||
|
`narratio config diff <left-profile> <right-profile>` uses the same pipeline and
|
||||||
|
campaign selectors, resolves each named profile independently from one parsed
|
||||||
|
root source set, and does not accept a separate `--profile` flag.
|
||||||
|
|
||||||
|
Campaign selection is optional only when the resolved pipeline has no
|
||||||
|
`scriptorium.artifact_families`. When families are declared, Narratio selects a
|
||||||
|
campaign through an explicit flag or `pipeline.campaigns.default_campaign_id`,
|
||||||
|
then parses the campaign-owned party and expands concrete artifacts and any
|
||||||
|
family publish rules before validation. `config validate` prints the resulting
|
||||||
|
root, profile, and effective digest. `config show` emits the normalized
|
||||||
|
effective pipeline YAML, with defaults and concrete expansion included but
|
||||||
|
composition and family declarations omitted. `config sources` prints a stable
|
||||||
|
source projection instead of effective values: root/import/profile/default
|
||||||
|
ownership plus campaign/party and generated-family records. Canonical derived
|
||||||
|
players trace to the party; a legacy configured players file is explicitly
|
||||||
|
marked as a legacy player source. The [CLI reference](cli.md#config-validate-config-show-config-sources-and-config-diff)
|
||||||
|
owns command syntax and output conventions.
|
||||||
|
|
||||||
|
`config diff` compares normalized field values rather than YAML text or source
|
||||||
|
ownership. It emits sorted `added`, `removed`, and `changed` records, uses
|
||||||
|
compact deterministic JSON values, treats lists atomically, and reports `no
|
||||||
|
differences` when the complete effective configurations are equal. Concrete
|
||||||
|
family members and generated publish rules participate after expansion; moving
|
||||||
|
an equal value between eligible root/import sources does not create a
|
||||||
|
difference.
|
||||||
|
|
||||||
|
### Migrating to the maintained bundle
|
||||||
|
|
||||||
|
Use the [production/testing bundle](../examples/production-testing/pipeline.yml)
|
||||||
|
as the complete copyable migration reference. Split stable pipeline settings
|
||||||
|
into explicit additive imports, place production/testing differences in one
|
||||||
|
selected overlay, and retain a production `default_profile`. Convert campaign
|
||||||
|
rosters to [canonical party input](integrations/party.md), remove a separate
|
||||||
|
`players_file`, then express character work as families. Inspect the result
|
||||||
|
with `config validate`, `config show`, and `config sources`; use `config diff`
|
||||||
|
to review profiles before running a session. Unversioned parties and their
|
||||||
|
`players_file` remain a clearly bounded legacy compatibility path.
|
||||||
|
|
||||||
|
### Identity segments
|
||||||
|
|
||||||
|
Campaign IDs (`campaign_id` and `default_campaign_id`), session IDs, previous
|
||||||
|
session IDs, and Narratio run IDs are opaque portable segments. They must use
|
||||||
|
only ASCII letters, digits, `.`, `_`, and `-`; empty values, `.`/`..`, path
|
||||||
|
separators, drive forms, whitespace, control characters, and non-ASCII text are
|
||||||
|
rejected. Narratio does not trim or rewrite these values. Existing manifests or
|
||||||
|
remote state with an unsafe legacy identity must be migrated before use.
|
||||||
|
|
||||||
|
## Validation and Merge Rules
|
||||||
|
|
||||||
|
- YAML decode is strict (`KnownFields(true)`) and accepts exactly one document:
|
||||||
|
unknown fields or trailing documents fail load.
|
||||||
|
- A pipeline file may explicitly import additive YAML fragments through the
|
||||||
|
root-only `composition.imports` list. Imported files contribute fields to one
|
||||||
|
logical pipeline document; they do not override fields supplied by the root
|
||||||
|
or another import.
|
||||||
|
- A root pipeline may declare named profiles. Exactly one profile is selected
|
||||||
|
by an option-aware caller or by `composition.default_profile`; a caller's
|
||||||
|
explicit selection takes precedence. Declaring profiles without either form
|
||||||
|
of selection is an error.
|
||||||
|
- Configured timeout and retry-delay durations must be positive. An omitted
|
||||||
|
artifact timeout continues to inherit its configured Scriptorium timeout.
|
||||||
|
- Session files must be concrete; unresolved `{{ ... }}` placeholders fail load.
|
||||||
|
- Pipeline defaults are applied before validation.
|
||||||
|
- Campaign and session identities must agree.
|
||||||
|
- Required stable files (`speakers_file`, `autocorrect_file`, `glossary_file`,
|
||||||
|
`party_file`) and the optional `spell_catalog_file` resolve from session
|
||||||
|
overrides when provided, otherwise from campaign defaults. An empty or
|
||||||
|
omitted session spell-catalog value inherits the campaign value.
|
||||||
|
- `party_file` is classified when pipeline and campaign configuration are
|
||||||
|
combined. A versioned [canonical party](integrations/party.md) is
|
||||||
|
campaign-owned, derives the players input internally, and forbids both a
|
||||||
|
separate `players_file` and a session `party_file` override. An unversioned
|
||||||
|
party remains a bounded legacy input and requires `players_file`; its normal
|
||||||
|
campaign/session overrides continue to apply.
|
||||||
|
- Exactly one audio mode must be configured in session input:
|
||||||
|
- local (`audio_dir` or `audio_files`), or
|
||||||
|
- S3 (`audio_s3.prefix`).
|
||||||
|
|
||||||
|
### Pipeline composition
|
||||||
|
|
||||||
|
Large pipeline configurations may be split into explicitly named fragments and
|
||||||
|
may declare one overlay per selectable profile:
|
||||||
|
|
||||||
```yaml
|
```yaml
|
||||||
whisperx:
|
composition:
|
||||||
transcribe_url: "https://transcription.example.com/transcribe"
|
imports:
|
||||||
|
- config/storage.yml
|
||||||
|
- config/integrations.yaml
|
||||||
|
default_profile: production
|
||||||
|
profiles:
|
||||||
|
production:
|
||||||
|
overlay: profiles/production.yml
|
||||||
|
testing:
|
||||||
|
overlay: profiles/testing.yml
|
||||||
|
|
||||||
|
campaigns:
|
||||||
|
root: /usr/local/share/narratio/campaigns
|
||||||
```
|
```
|
||||||
|
|
||||||
Why this is sufficient:
|
The maintained [production/testing bundle](../examples/production-testing/pipeline.yml)
|
||||||
|
is a complete copyable example of this structure, including canonical-party
|
||||||
|
artifact families.
|
||||||
|
|
||||||
- `whisperx.transcribe_url` is required.
|
Imports are resolved relative to the directory containing the root pipeline
|
||||||
- `workspace.root` defaults to `/var/lib/narratio`.
|
file and are loaded in declaration order. Narratio does not scan directories or
|
||||||
- optional sections (`seriatim`, `audita`, `archive`, `scriptorium`, `trim`, `normalize`, etc.) receive defaults or stay inactive.
|
infer fragments. Each import must be a confined regular `.yml` or `.yaml` file:
|
||||||
|
absolute paths, traversal, symlinks, directories, duplicate files, and an
|
||||||
|
import of the root pipeline itself are rejected. Only the root pipeline may
|
||||||
|
contain `composition`; nested composition is rejected.
|
||||||
|
|
||||||
## 5. Minimal session template
|
Composition is additive. A map may be extended by multiple files when every
|
||||||
|
leaf is distinct, but a scalar, list, or map/list/scalar kind cannot be claimed
|
||||||
|
more than once, even when the repeated values are identical. Conflict errors
|
||||||
|
name the full field path and every source that claimed it. The assembled YAML is
|
||||||
|
then decoded against the normal strict pipeline schema and defaults are applied
|
||||||
|
once.
|
||||||
|
|
||||||
|
Profile names are case-sensitive, non-empty, trimmed, and cannot contain
|
||||||
|
control characters. If `profiles` is present, it must contain at least one
|
||||||
|
entry and every entry must contain only an `overlay` path. An explicit profile
|
||||||
|
selection overrides `default_profile`; unknown and explicitly empty selections
|
||||||
|
fail. Narratio never selects the first profile implicitly and does not read a
|
||||||
|
profile selection from the environment.
|
||||||
|
|
||||||
|
Every declared overlay is resolved relative to the root pipeline directory and
|
||||||
|
must satisfy the same confined regular-YAML-file rules as an import. Narratio
|
||||||
|
parses every declared overlay even when it is not selected, then applies only
|
||||||
|
the selected one. Maps merge recursively, overlay scalars replace base scalars,
|
||||||
|
and overlay lists replace base lists completely. Explicit `false`, zero, empty
|
||||||
|
lists, and empty maps remain meaningful. YAML null cannot delete a value, and
|
||||||
|
kind changes are rejected. Profiles cannot inherit from or stack with other
|
||||||
|
profiles, and overlays cannot import files or declare profiles.
|
||||||
|
|
||||||
|
After composition, Narratio strictly decodes the result, applies centralized
|
||||||
|
defaults once, resolves ordinary paths, and computes a deterministic effective
|
||||||
|
configuration digest. The digest represents the normalized, secret-free
|
||||||
|
runtime pipeline mapping; it excludes composition declarations, source
|
||||||
|
provenance, profile identity, and raw environment secret values. Equivalent
|
||||||
|
effective mappings therefore have the same digest regardless of how fields are
|
||||||
|
split among the root and imports.
|
||||||
|
|
||||||
|
An imported field has the same meaning it would have in a monolithic root
|
||||||
|
pipeline. In particular, ordinary relative pipeline paths continue to resolve
|
||||||
|
from the root pipeline directory, not from the importing fragment's directory.
|
||||||
|
|
||||||
|
## Minimal Working Configuration
|
||||||
|
|
||||||
|
`pipeline.yml`
|
||||||
|
|
||||||
```yaml
|
```yaml
|
||||||
session_id: "{{ session_id }}"
|
campaigns:
|
||||||
campaign: sample-campaign
|
root: /usr/local/share/narratio/campaigns
|
||||||
|
default_campaign_id: sample-campaign
|
||||||
|
whisperx:
|
||||||
|
transcribe_url: https://transcription.example.com/transcribe
|
||||||
|
```
|
||||||
|
|
||||||
|
`campaign.yml`
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
campaign_id: sample-campaign
|
||||||
|
inputs:
|
||||||
|
speakers_file: ./speakers.yml
|
||||||
|
autocorrect_file: ./autocorrect.yml
|
||||||
|
glossary_file: ./glossary.yml
|
||||||
|
players_file: ./players.yml
|
||||||
|
party_file: ./party.yml
|
||||||
|
```
|
||||||
|
|
||||||
|
`session.yml` (local audio)
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
session_id: 2026-05-03
|
||||||
inputs:
|
inputs:
|
||||||
audio_dir: ./audio
|
audio_dir: ./audio
|
||||||
speakers_file: ./examples/speakers.yml
|
|
||||||
autocorrect_file: ./examples/autocorrect.yml
|
|
||||||
glossary_file: ./examples/glossary.yml
|
|
||||||
```
|
```
|
||||||
|
|
||||||
Usage:
|
## Secrets Handling
|
||||||
|
|
||||||
```bash
|
- Do not place raw secrets in YAML.
|
||||||
narratio run --config /path/to/pipeline.yml --session ./session.yml --session-id 2026-05-03
|
- Use env var names in config (for example `pipeline.audita.llm_api_key_env`).
|
||||||
```
|
- Optionally load credential files from `pipeline.secrets.env_dir`. Each valid
|
||||||
|
environment-variable filename supplies one value; trailing CR/LF is removed.
|
||||||
|
- An existing process environment value takes precedence over a credential file.
|
||||||
|
- Credential directories and files must not be symlinks and must be regular,
|
||||||
|
bounded files (at most 8 KiB per value). On POSIX, provision the directory
|
||||||
|
with no group/other access (normally `0700`) and files with no group/other
|
||||||
|
access (normally `0600`).
|
||||||
|
- Commands that need storage/auth load filesystem secrets before constructing adapters.
|
||||||
|
|
||||||
## 6. Production-oriented config
|
## Publish Configuration Summary
|
||||||
|
|
||||||
|
Publish rules live under `pipeline.publish`.
|
||||||
|
|
||||||
```yaml
|
```yaml
|
||||||
workspace:
|
publish:
|
||||||
root: /var/lib/narratio/workspace
|
|
||||||
cleanup_after_archive: true
|
|
||||||
|
|
||||||
storage:
|
|
||||||
backend: s3
|
|
||||||
s3:
|
|
||||||
bucket: my-dnd-archive
|
|
||||||
root_prefix: dnd
|
|
||||||
region: us-east-1
|
|
||||||
access_key_id_env: OBJECT_STORAGE_KEY_ID
|
|
||||||
secret_access_key_env: OBJECT_STORAGE_KEY
|
|
||||||
|
|
||||||
spool:
|
|
||||||
root: /var/spool/narratio
|
|
||||||
delete_audio_after_archive: true
|
|
||||||
|
|
||||||
archive:
|
|
||||||
enabled: true
|
enabled: true
|
||||||
upload_run: true
|
upload_run: true
|
||||||
promote_artifacts:
|
outputs:
|
||||||
- from: transcripts/trimmed.json
|
- source: narratio.transcript.final_trimmed
|
||||||
to: transcripts/trimmed.json
|
dest: transcripts/final.trimmed.json
|
||||||
required: true
|
required: true
|
||||||
- from: artifacts/session_recap.md
|
- source: narratio.transcript.final_markdown
|
||||||
to: artifacts/session_recap.md
|
dest: transcripts/final.md
|
||||||
required: true
|
required: true
|
||||||
|
- source: narratio.transcript.final_trimmed_markdown
|
||||||
whisperx:
|
dest: transcripts/final.trimmed.md
|
||||||
transcribe_url: "https://transcription.example.com/transcribe"
|
required: true
|
||||||
|
- source: narratio.artifact.session_recap
|
||||||
scriptorium:
|
dest: artifacts/session_recap.md
|
||||||
artifacts:
|
required: true
|
||||||
session_recap:
|
locks:
|
||||||
enabled: true
|
- source: narratio.artifact.session_recap
|
||||||
prompt_id: dnd.session_recap
|
reason: manual post-publish edits
|
||||||
output_path: artifacts/session_recap.md
|
|
||||||
inputs:
|
|
||||||
transcript:
|
|
||||||
source: narratio.transcript.trimmed
|
|
||||||
required: true
|
|
||||||
```
|
```
|
||||||
|
|
||||||
Operational notes:
|
Rules:
|
||||||
|
|
||||||
- archive promotion is explicit and path-based via `archive.promote_artifacts`.
|
- `outputs[].source` is required.
|
||||||
- Narratio does not auto-promote all generated analyze artifacts.
|
- `outputs[].dest` may be omitted when derivable from source.
|
||||||
|
- extraction sources require an explicit `outputs[].dest` and publish only when
|
||||||
|
a rule names that source; the Notarius index and complete bundle are not
|
||||||
|
publish sources.
|
||||||
|
- `outputs[].required` defaults to `true`.
|
||||||
|
- static locks (`pipeline.publish.locks`) merge with remote locks (`{session_prefix}/locks.yml`), with static locks taking precedence on duplicates.
|
||||||
|
|
||||||
## 7. Full pipeline reference
|
## Full Schema
|
||||||
|
|
||||||
| Path | Type | Required | Default |
|
### Pipeline
|
||||||
|
|
||||||
|
| Field | Type | Required | Default / Rule |
|
||||||
| --- | --- | --- | --- |
|
| --- | --- | --- | --- |
|
||||||
|
| `composition.imports[]` | list of strings | No | explicit additive pipeline fragments relative to the root pipeline directory; `.yml` or `.yaml` regular files only |
|
||||||
|
| `composition.default_profile` | string | Conditional | selected when profiles exist and no caller explicitly selects one; must name a declared profile |
|
||||||
|
| `composition.profiles.<name>.overlay` | string | Conditional | required for every declared profile; one confined `.yml` or `.yaml` overlay relative to the root pipeline directory |
|
||||||
| `pipeline.workspace.root` | string | No | `/var/lib/narratio` |
|
| `pipeline.workspace.root` | string | No | `/var/lib/narratio` |
|
||||||
| `pipeline.workspace.cleanup_after_archive` | bool | No | `false` |
|
| `pipeline.workspace.cleanup_after_publish` | bool | No | `false` |
|
||||||
| `pipeline.secrets.env_dir` | string | Conditional | none |
|
| `pipeline.campaigns.root` | string | No | `/usr/local/share/narratio/campaigns` |
|
||||||
| `pipeline.storage.backend` | string | No | empty |
|
| `pipeline.campaigns.default_campaign_id` | string | No | empty |
|
||||||
| `pipeline.storage.bucket` | string | No | empty |
|
| `pipeline.secrets.env_dir` | string | No | empty |
|
||||||
| `pipeline.storage.prefix` | string | No | empty |
|
| `pipeline.storage.backend` | string | No | `local`; supported values are `local` and `s3` (case-insensitive) |
|
||||||
| `pipeline.storage.s3.bucket` | string | Conditional | empty |
|
| `pipeline.storage.s3.bucket` | string | Conditional | required when backend is `s3` and S3 session-audio or publish upload is enabled |
|
||||||
| `pipeline.storage.s3.root_prefix` | string | No | `dnd` |
|
| `pipeline.storage.s3.root_prefix` | string | No | `dnd` |
|
||||||
| `pipeline.storage.s3.region` | string | No | empty |
|
| `pipeline.storage.s3.region` | string | No | empty |
|
||||||
| `pipeline.storage.s3.endpoint` | string | No | empty |
|
| `pipeline.storage.s3.endpoint` | string | No | empty |
|
||||||
@@ -151,21 +297,26 @@ Operational notes:
|
|||||||
| `pipeline.storage.s3.access_key_id_env` | string | No | `OBJECT_STORAGE_KEY_ID` |
|
| `pipeline.storage.s3.access_key_id_env` | string | No | `OBJECT_STORAGE_KEY_ID` |
|
||||||
| `pipeline.storage.s3.secret_access_key_env` | string | No | `OBJECT_STORAGE_KEY` |
|
| `pipeline.storage.s3.secret_access_key_env` | string | No | `OBJECT_STORAGE_KEY` |
|
||||||
| `pipeline.spool.root` | string | No | `/var/spool/narratio` |
|
| `pipeline.spool.root` | string | No | `/var/spool/narratio` |
|
||||||
| `pipeline.spool.delete_audio_after_archive` | bool | No | `false` |
|
| `pipeline.spool.delete_audio_after_publish` | bool | No | `false` |
|
||||||
| `pipeline.archive.enabled` | bool | No | `true` |
|
| `pipeline.cache.root` | string | No | `/var/cache/narratio` |
|
||||||
| `pipeline.archive.upload_run` | bool | No | `true` |
|
| `pipeline.cache.s3_audio` | bool | No | `true` |
|
||||||
| `pipeline.archive.promote_artifacts[]` | list | No | trimmed + session_recap rules |
|
| `pipeline.publish.enabled` | bool | No | `true` |
|
||||||
| `pipeline.archive.promote_artifacts[].from` | string | Yes (per rule) | none |
|
| `pipeline.publish.upload_run` | bool | No | `true` |
|
||||||
| `pipeline.archive.promote_artifacts[].to` | string | Yes (per rule) | none |
|
| `pipeline.publish.outputs[]` | list | No | defaults to final trimmed JSON plus final and final-trimmed Markdown outputs |
|
||||||
| `pipeline.archive.promote_artifacts[].required` | bool | No | `true` |
|
| `pipeline.publish.outputs[].source` | string | Yes (per rule) | must reference built-in or configured artifact source |
|
||||||
| `pipeline.whisperx.transcribe_url` | string | Yes | none |
|
| `pipeline.publish.outputs[].dest` | string | Conditional | derived if omitted and source supports derivation |
|
||||||
|
| `pipeline.publish.outputs[].required` | bool | No | `true` |
|
||||||
|
| `pipeline.publish.locks[]` | list | No | empty |
|
||||||
|
| `pipeline.publish.locks[].source` | string | Yes (per lock) | must reference supported publish source |
|
||||||
|
| `pipeline.publish.locks[].reason` | string | No | empty |
|
||||||
|
| `pipeline.whisperx.transcribe_url` | string | Yes | absolute `http` or `https` URL |
|
||||||
| `pipeline.whisperx.language` | string | No | `en` |
|
| `pipeline.whisperx.language` | string | No | `en` |
|
||||||
| `pipeline.whisperx.timeout` | duration string | No | `30m` |
|
| `pipeline.whisperx.timeout` | duration | No | `30m` |
|
||||||
| `pipeline.whisperx.retries` | int | No | `3` |
|
| `pipeline.whisperx.retries` | int | No | `3` |
|
||||||
| `pipeline.whisperx.retry_delay` | duration string | No | `2s` |
|
| `pipeline.whisperx.retry_delay` | duration | No | `2s` |
|
||||||
| `pipeline.whisperx.concurrency` | int | No | `2` |
|
| `pipeline.whisperx.concurrency` | int | No | `2` |
|
||||||
| `pipeline.seriatim.binary` | string | No | `seriatim` |
|
| `pipeline.seriatim.binary` | string | No | `seriatim` |
|
||||||
| `pipeline.seriatim.timeout` | duration string | No | `10m` |
|
| `pipeline.seriatim.timeout` | duration | No | `10m` |
|
||||||
| `pipeline.seriatim.output_schema` | string | No | `seriatim-intermediate` |
|
| `pipeline.seriatim.output_schema` | string | No | `seriatim-intermediate` |
|
||||||
| `pipeline.seriatim.coalesce_gap` | float | No | `3.0` |
|
| `pipeline.seriatim.coalesce_gap` | float | No | `3.0` |
|
||||||
| `pipeline.seriatim.report` | bool | No | `true` |
|
| `pipeline.seriatim.report` | bool | No | `true` |
|
||||||
@@ -174,7 +325,7 @@ Operational notes:
|
|||||||
| `pipeline.seriatim.env.backchannel_max_duration` | float | No | unset |
|
| `pipeline.seriatim.env.backchannel_max_duration` | float | No | unset |
|
||||||
| `pipeline.seriatim.env.filler_max_duration` | float | No | unset |
|
| `pipeline.seriatim.env.filler_max_duration` | float | No | unset |
|
||||||
| `pipeline.audita.binary` | string | No | `audita` |
|
| `pipeline.audita.binary` | string | No | `audita` |
|
||||||
| `pipeline.audita.timeout` | duration string | No | `3h` |
|
| `pipeline.audita.timeout` | duration | No | `3h` |
|
||||||
| `pipeline.audita.llm_api_key_env` | string | No | empty |
|
| `pipeline.audita.llm_api_key_env` | string | No | empty |
|
||||||
| `pipeline.audita.modules[]` | list[string] | No | empty |
|
| `pipeline.audita.modules[]` | list[string] | No | empty |
|
||||||
| `pipeline.audita.base_url` | string | No | empty |
|
| `pipeline.audita.base_url` | string | No | empty |
|
||||||
@@ -188,117 +339,234 @@ Operational notes:
|
|||||||
| `pipeline.audita.output_schema` | string | No | empty |
|
| `pipeline.audita.output_schema` | string | No | empty |
|
||||||
| `pipeline.audita.work_dir_retention` | string | No | empty |
|
| `pipeline.audita.work_dir_retention` | string | No | empty |
|
||||||
| `pipeline.audita.report` | bool | No | `true` |
|
| `pipeline.audita.report` | bool | No | `true` |
|
||||||
| `pipeline.normalize.output_path` | string | No | `transcripts/normalized.json` |
|
| `pipeline.normalize.output_path` | string | No | `transcripts/final.json` |
|
||||||
| `pipeline.normalize.output_schema` | string | No | `seriatim-intermediate` |
|
| `pipeline.normalize.output_schema` | string | No | `seriatim-intermediate` |
|
||||||
| `pipeline.normalize.report` | bool | No | `true` |
|
| `pipeline.normalize.report` | bool | No | `true` |
|
||||||
| `pipeline.trim.enabled` | bool | No | `false` |
|
| `pipeline.trim.enabled` | bool | No | `true` |
|
||||||
| `pipeline.trim.output_path` | string | Conditional | none |
|
| `pipeline.trim.output_path` | string | No | `transcripts/final.trimmed.json` |
|
||||||
| `pipeline.trim.bounds.prompt_id` | string | Conditional | none |
|
| `pipeline.trim.bounds.prompt_id` | string | No | `dnd.session_bounds` |
|
||||||
| `pipeline.trim.bounds.profile_id` | string | No | empty |
|
| `pipeline.trim.bounds.profile_id` | string | No | empty |
|
||||||
| `pipeline.trim.bounds.transcript_input_name` | string | Conditional | none |
|
| `pipeline.trim.bounds.transcript_input_name` | string | No | `transcript` |
|
||||||
| `pipeline.trim.bounds.output_path` | string | Conditional | none |
|
| `pipeline.trim.bounds.output_path` | string | No | `artifacts/session_bounds.json` |
|
||||||
| `pipeline.trim.bounds.timeout` | duration string | No | `10m` |
|
| `pipeline.trim.bounds.timeout` | duration | No | `10m` |
|
||||||
| `pipeline.trim.bounds.render_debug` | bool | No | `false` |
|
| `pipeline.trim.bounds.render_debug` | bool | No | `false` |
|
||||||
| `pipeline.trim.bounds.render_output_path` | string | Conditional | none |
|
| `pipeline.trim.bounds.render_output_path` | string | Conditional | required when `render_debug` is true |
|
||||||
| `pipeline.trim.seriatim.report` | bool | No | `false` |
|
| `pipeline.trim.seriatim.report` | bool | No | `false` |
|
||||||
|
| `pipeline.notarius.enabled` | bool | No | `false` |
|
||||||
|
| `pipeline.notarius.binary` | string | No | `notarius` |
|
||||||
|
| `pipeline.notarius.config_path` | string | Conditional | required when enabled; relative paths resolve from the pipeline file directory |
|
||||||
|
| `pipeline.notarius.pipeline_id` | string | Conditional | required when enabled |
|
||||||
|
| `pipeline.notarius.timeout` | duration | No | `3h`; must be positive |
|
||||||
|
| `pipeline.notarius.working_directory` | string | No | directory containing resolved `config_path`; relative paths resolve from the pipeline file directory |
|
||||||
|
| `pipeline.notarius.references` | map[string]string | No | empty; maps normalized Notarius selectors to supported prepared Narratio source IDs; maximum 256 entries |
|
||||||
|
| `pipeline.notarius.outputs` | map | Conditional | at least one entry when enabled |
|
||||||
|
| `pipeline.render.enabled` | bool | No | `true` |
|
||||||
|
| `pipeline.render.format` | string | No | `markdown` (only supported value) |
|
||||||
|
| `pipeline.render.title` | string | No | empty (falls back to `session.title` when set) |
|
||||||
|
| `pipeline.render.include_timestamps` | bool | No | `true` |
|
||||||
|
| `pipeline.render.include_segment_ids` | bool | No | `true` |
|
||||||
|
| `pipeline.render.include_metadata` | bool | No | `false` |
|
||||||
| `pipeline.scriptorium.binary` | string | No | `scriptorium` |
|
| `pipeline.scriptorium.binary` | string | No | `scriptorium` |
|
||||||
| `pipeline.scriptorium.config_path` | string | No | empty |
|
| `pipeline.scriptorium.config_path` | string | No | empty |
|
||||||
| `pipeline.scriptorium.timeout` | duration string | No | `10m` |
|
| `pipeline.scriptorium.timeout` | duration | No | `10m` |
|
||||||
| `pipeline.scriptorium.render_debug` | bool | No | `false` |
|
| `pipeline.scriptorium.render_debug` | bool | No | `false` |
|
||||||
| `pipeline.scriptorium.artifacts` | map | No | empty |
|
| `pipeline.scriptorium.artifacts` | map | No | empty |
|
||||||
| `pipeline.scriptorium.artifacts.<name>.enabled` | bool | No | `false` |
|
| `pipeline.scriptorium.artifact_families` | map | No | empty; expands one ordinary artifact per canonical party character |
|
||||||
| `pipeline.scriptorium.artifacts.<name>.depends_on[]` | list[string] | No | empty |
|
| `pipeline.notification.mode` | string | No | `noop`; the only supported notification mode until a provider is implemented |
|
||||||
| `pipeline.scriptorium.artifacts.<name>.render_debug` | bool | No | unset |
|
|
||||||
| `pipeline.scriptorium.artifacts.<name>.prompt_id` | string | Conditional | none |
|
|
||||||
| `pipeline.scriptorium.artifacts.<name>.profile_id` | string | No | empty |
|
|
||||||
| `pipeline.scriptorium.artifacts.<name>.output_path` | string | Conditional | none |
|
|
||||||
| `pipeline.scriptorium.artifacts.<name>.timeout` | duration string | No | empty |
|
|
||||||
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.source` | string | Conditional | none |
|
|
||||||
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.artifact` | string | No | empty |
|
|
||||||
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.path` | string | No | empty |
|
|
||||||
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.required` | bool | No | `false` |
|
|
||||||
| `pipeline.scriptorium.artifacts.<name>.vars.<key>` | map value | No | empty |
|
|
||||||
| `pipeline.analyzer.binary_path` | string | No | empty |
|
|
||||||
| `pipeline.analyzer.timeout` | duration string | No | empty |
|
|
||||||
| `pipeline.analyzer.artifacts.output_dir` | string | No | empty |
|
|
||||||
| `pipeline.analyzer.artifacts.types[]` | list[string] | No | empty |
|
|
||||||
| `pipeline.notification.backend` | string | No | empty |
|
|
||||||
| `pipeline.notification.recipient` | string | No | empty |
|
|
||||||
| `pipeline.notification.timeout` | duration string | No | empty |
|
|
||||||
|
|
||||||
Scriptorium artifact-key and dependency rules:
|
### Notarius Reference Bindings
|
||||||
|
|
||||||
- artifact keys must match `^[a-z][a-z0-9_]*$`.
|
`pipeline.notarius.references` maps a Notarius CLI selector to a prepared
|
||||||
- enabled artifacts require `prompt_id` and `output_path`.
|
Narratio source, not to a filesystem path:
|
||||||
- `output_path` must be relative, traversal-safe, and under `artifacts/`.
|
|
||||||
- configured artifact input sources use `narratio.artifact.<name>`.
|
|
||||||
- if input source references `narratio.artifact.<name>`, artifact `<name>` must exist and must be listed in `depends_on`.
|
|
||||||
- every `depends_on` entry must be a configured artifact key.
|
|
||||||
- self-dependency is rejected.
|
|
||||||
- enabled dependency cycles are rejected.
|
|
||||||
- any artifact referenced by `depends_on` or `narratio.artifact.<name>` source must define `output_path` (even if not enabled).
|
|
||||||
|
|
||||||
Allowed `pipeline.scriptorium.artifacts.<name>.inputs.<key>.source` values:
|
```yaml
|
||||||
|
notarius:
|
||||||
|
references:
|
||||||
|
glossary: narratio.input.glossary
|
||||||
|
party: narratio.input.party
|
||||||
|
players: narratio.input.players
|
||||||
|
spell_catalog: narratio.input.spell_catalog
|
||||||
|
```
|
||||||
|
|
||||||
- `previous_session_artifact`
|
Supported sources are `narratio.input.party`, `narratio.input.players`,
|
||||||
- `narratio.transcript.merged`
|
`narratio.input.glossary`, and `narratio.input.spell_catalog`. Each map entry is
|
||||||
- `narratio.transcript.polished`
|
required by its presence: omit a binding when the selected Notarius pipeline
|
||||||
- `narratio.transcript.full`
|
does not need it. A spell-catalog binding additionally requires an effective
|
||||||
- `narratio.transcript.trimmed`
|
campaign or session `spell_catalog_file`.
|
||||||
- `narratio.bounds.session`
|
|
||||||
- `narratio.artifact.<configured_artifact_key>`
|
|
||||||
|
|
||||||
## 8. Full session reference
|
Selectors accept Notarius's `slot`, `chunk.slot`, `lane.slot`,
|
||||||
|
`lane.extract.slot`, `lane.merge.slot`, and `lane.normalize.slot` forms.
|
||||||
|
Narratio trims whitespace around
|
||||||
|
selectors and their dot-separated components, rejects empty components and
|
||||||
|
`=`, rejects duplicate normalized selectors, and limits the map to 256 entries.
|
||||||
|
It validates only selector structure and the prepared source vocabulary;
|
||||||
|
Notarius owns target-slot declarations and media compatibility.
|
||||||
|
|
||||||
| Path | Type | Required | Default |
|
Before extraction, Narratio resolves every binding from the current prepared
|
||||||
|
session manifest and streams it into a verified invocation-local snapshot whose
|
||||||
|
absolute path is passed to Notarius. Missing, unsafe, empty,
|
||||||
|
changed-during-copy, or checksum-inconsistent prepared evidence fails with
|
||||||
|
guidance to force `prepare`. Bindings are sorted by normalized selector and are
|
||||||
|
part of extraction fingerprint and resume identity. See the
|
||||||
|
[Notarius integration contract](./integrations/notarius.md) for the subprocess
|
||||||
|
boundary and the [complete example](../examples/pipeline.full.annotated.yml)
|
||||||
|
for a copyable configuration.
|
||||||
|
|
||||||
|
### Notarius Output Entries
|
||||||
|
|
||||||
|
For each `pipeline.notarius.outputs.<name>`:
|
||||||
|
|
||||||
|
| Field | Type | Required | Rule |
|
||||||
| --- | --- | --- | --- |
|
| --- | --- | --- | --- |
|
||||||
| `session.session_id` | string | Yes | none |
|
| `lane_id` | string | Yes | unique Notarius lane ID |
|
||||||
| `session.campaign` | string | Yes | none |
|
| `media_type` | string | Yes | exact accepted descriptor media type |
|
||||||
| `session.date` | string | No | empty |
|
| `schema_id` | string | Yes | exact accepted descriptor schema ID |
|
||||||
| `session.title` | string | No | empty |
|
| `schema_version` | string | Yes | exact accepted descriptor schema version |
|
||||||
| `session.inputs.audio_dir` | string | Conditional | empty |
|
| `module_key` | string | No | exact accepted module key when set |
|
||||||
| `session.inputs.audio_files[]` | list[string] | Conditional | empty |
|
|
||||||
| `session.inputs.audio_s3.prefix` | string | Conditional | none |
|
|
||||||
| `session.inputs.speakers_file` | string | Yes | none |
|
|
||||||
| `session.inputs.autocorrect_file` | string | Yes | none |
|
|
||||||
| `session.inputs.glossary_file` | string | Yes | none |
|
|
||||||
|
|
||||||
Audio-source rule:
|
Output names must match `^[a-z][a-z0-9_]*$` and become selectable sources named
|
||||||
|
`narratio.extraction.<name>`. Lane IDs must be unique. Every declared output is
|
||||||
|
required from a successful Notarius result; a missing, rejected, duplicate, or
|
||||||
|
contract-incompatible lane fails extraction. See the
|
||||||
|
[complete maintained example](../examples/pipeline.full.annotated.yml) for the
|
||||||
|
current ten-lane D&D mapping and the [Notarius contract](./integrations/notarius.md)
|
||||||
|
for compatibility ownership.
|
||||||
|
|
||||||
- configure exactly one mode:
|
### Scriptorium Artifact Entries
|
||||||
- `audio_dir`, or
|
|
||||||
- `audio_files` (at least one), or
|
|
||||||
- `audio_s3.prefix`
|
|
||||||
- `audio_s3` cannot be combined with local audio fields.
|
|
||||||
|
|
||||||
## 9. Secrets
|
For each `pipeline.scriptorium.artifacts.<name>`:
|
||||||
|
|
||||||
Narratio supports filesystem-based secret injection via `pipeline.secrets.env_dir`.
|
| Field | Type | Required | Rule |
|
||||||
|
| --- | --- | --- | --- |
|
||||||
|
| `enabled` | bool | No | `false` if omitted |
|
||||||
|
| `depends_on[]` | list[string] | No | must reference configured artifact keys; no self-reference; configured graph must be acyclic |
|
||||||
|
| `render_debug` | bool | No | per-artifact override |
|
||||||
|
| `prompt_id` | string | Conditional | required when artifact is enabled |
|
||||||
|
| `profile_id` | string | No | empty |
|
||||||
|
| `output_path` | string | Conditional | required when enabled; also required when referenced by publish/output/input rules |
|
||||||
|
| `timeout` | duration | No | artifact override |
|
||||||
|
| `inputs` | map | No | input key names must be non-empty |
|
||||||
|
| `vars` | map | No | values must be string or bool; `session_id` is reserved and overwritten by Narratio |
|
||||||
|
|
||||||
Behavior:
|
Narratio adds `session_id=narratio-session-<session_id>` to every Scriptorium request for sticky upstream LLM routing. If an artifact config sets `vars.session_id`, Narratio replaces that value before invoking Scriptorium. Use a different variable name if a prompt needs the raw Narratio session ID as content.
|
||||||
|
|
||||||
- `env_dir` may be absolute or relative.
|
Without `--artifacts`, analyze executes enabled configured artifacts. With an
|
||||||
- relative `env_dir` resolves from current working directory.
|
explicit `--artifacts` list, the exact named configured artifacts are the
|
||||||
- files with valid env-var names (`[A-Za-z_][A-Za-z0-9_]*`) are loaded.
|
one-invocation targets even if their `enabled` values are false. Analyze closes
|
||||||
- values are loaded from file contents with trailing newline trimming.
|
those targets over `depends_on`: a current prerequisite is reused, while a
|
||||||
- existing process env vars are preserved.
|
stale, missing, failed, or legacy prerequisite is rebuilt before its dependent.
|
||||||
- invalid names and subdirectories are skipped.
|
Unrelated artifacts are not executed. Named targets and any prerequisite that
|
||||||
- missing/unreadable `env_dir` fails command execution.
|
may require rebuilding must therefore have valid executable fields. This
|
||||||
|
override affects analyze planning only; publish uses the list only to filter
|
||||||
|
configured `narratio.artifact.<name>` output rules.
|
||||||
|
|
||||||
Guidance:
|
For each artifact input `pipeline.scriptorium.artifacts.<name>.inputs.<input_name>`:
|
||||||
|
|
||||||
- do not put secret values directly in YAML.
|
| Field | Type | Required | Rule |
|
||||||
- configure env var names in config and provide values via env/secrets files.
|
| --- | --- | --- | --- |
|
||||||
|
| `source` | string | Yes | built-in runtime source, prepared input source, `narratio.extraction.<name>`, `narratio.artifact.<name>`, or `narratio.previous_session.artifact.<name>` |
|
||||||
|
| `required` | bool | No | optional input requirement |
|
||||||
|
|
||||||
## 10. Examples
|
`artifact` and `path` are obsolete and rejected by strict configuration
|
||||||
|
loading. Use the canonical `source` identifier to select the input; Narratio
|
||||||
|
does not provide adapter-specific input passthrough fields.
|
||||||
|
|
||||||
Maintained examples:
|
### Scriptorium Artifact Families
|
||||||
|
|
||||||
- `examples/pipeline.minimal.yml`
|
`pipeline.scriptorium.artifact_families` declares a shared artifact template
|
||||||
- `examples/pipeline.production.yml`
|
for every canonical campaign character. Configuration resolution expands each
|
||||||
- `examples/pipeline.full.annotated.yml`
|
family into ordinary `pipeline.scriptorium.artifacts` entries before analyze
|
||||||
- `examples/session.template.yml`
|
planning or Scriptorium invocation. A legacy party cannot be used for a family.
|
||||||
- `examples/session.local-audio.yml`
|
|
||||||
- `examples/session.s3-audio.yml`
|
|
||||||
|
|
||||||
These examples are validated by `internal/config` tests.
|
For each `pipeline.scriptorium.artifact_families.<name>`:
|
||||||
|
|
||||||
|
| Field | Type | Required | Rule |
|
||||||
|
| --- | --- | --- | --- |
|
||||||
|
| `enabled`, `prompt_id`, `profile_id`, `timeout`, `render_debug`, `depends_on`, `inputs`, `vars` | ordinary artifact fields | No | copied to each generated artifact under the corresponding ordinary rules |
|
||||||
|
| `for_each` | string | Yes | exactly `party.characters` |
|
||||||
|
| `output_path_pattern` | string | Yes | safe path beneath `artifacts/` with exactly one `{character_id}` token and no other brace syntax |
|
||||||
|
| `member_vars` | map | No | maps an ordinary Scriptorium variable name to a supported canonical character selector |
|
||||||
|
| `member_dependencies` | list | No | unique family keys; each generated member depends on the corresponding generated member of each listed family |
|
||||||
|
| `publish` | map | No | typed family publish policy (`enabled`, `required`, `dest_pattern`) expanded into concrete publish outputs when enabled |
|
||||||
|
|
||||||
|
Generated keys are `<family>_<character_id>` and generated output paths must
|
||||||
|
not collide with explicit artifacts or another generated artifact. Families
|
||||||
|
expand even when disabled; normal analyze selection still omits disabled
|
||||||
|
artifacts unless they are explicitly selected by their concrete key.
|
||||||
|
|
||||||
|
Supported `member_vars` selectors are `character_id`, `player.name`,
|
||||||
|
`character.name`, `character.class_summary`, and `character.alias_summary`.
|
||||||
|
Their resolved values are strings. A member variable may not reuse a static
|
||||||
|
`vars` name; `session_id` remains owned and overwritten by Narratio as for any
|
||||||
|
other Scriptorium artifact.
|
||||||
|
|
||||||
|
Within a family only, an input source may use
|
||||||
|
`narratio.member_artifact.<family>`. The referenced family must be named in
|
||||||
|
that family's `member_dependencies`; resolution rewrites the source to the
|
||||||
|
corresponding ordinary `narratio.artifact.<family>_<character_id>` source.
|
||||||
|
This syntax is rejected in explicit artifacts and never reaches runtime stages
|
||||||
|
or Scriptorium.
|
||||||
|
|
||||||
|
### Notifications
|
||||||
|
|
||||||
|
Narratio currently supports only `notification.mode: noop`, which is also the
|
||||||
|
default when the section is omitted. The notify stage performs no delivery in
|
||||||
|
this mode. Backend, recipient, timeout, and other provider settings are
|
||||||
|
rejected by strict configuration loading until Narratio has a provider
|
||||||
|
integration.
|
||||||
|
|
||||||
|
### Campaign
|
||||||
|
|
||||||
|
| Field | Type | Required | Notes |
|
||||||
|
| --- | --- | --- | --- |
|
||||||
|
| `campaign_id` | string | Yes | canonical opaque campaign identity |
|
||||||
|
| `session_template_file` | string | No | used by `session init` when set |
|
||||||
|
| `inputs.speakers_file` | string | Yes | stable input default |
|
||||||
|
| `inputs.autocorrect_file` | string | Yes | stable input default |
|
||||||
|
| `inputs.glossary_file` | string | Yes | stable input default |
|
||||||
|
| `inputs.players_file` | string | Conditional | required only with an unversioned legacy `party_file`; forbidden for a canonical party |
|
||||||
|
| `inputs.party_file` | string | Yes | stable campaign party source; relative paths resolve from `campaign.yml` |
|
||||||
|
| `inputs.spell_catalog_file` | string | No | optional spell-catalog overlay default; required when a Notarius reference selects `narratio.input.spell_catalog` |
|
||||||
|
|
||||||
|
### Session
|
||||||
|
|
||||||
|
| Field | Type | Required in session file | Notes |
|
||||||
|
| --- | --- | --- | --- |
|
||||||
|
| `session_id` | string | Yes | opaque identity; must match CLI session target when provided |
|
||||||
|
| `previous_session_id` | string | No | opaque identity; must not equal `session_id` |
|
||||||
|
| `campaign` | string | No | opaque identity; filled from `campaign_id` during resolve if omitted |
|
||||||
|
| `date` | string | No | metadata |
|
||||||
|
| `title` | string | No | metadata |
|
||||||
|
| `inputs.speakers_file` | string | No | overrides campaign stable input |
|
||||||
|
| `inputs.autocorrect_file` | string | No | overrides campaign stable input |
|
||||||
|
| `inputs.glossary_file` | string | No | overrides campaign stable input |
|
||||||
|
| `inputs.players_file` | string | No | legacy-party override; forbidden for a canonical party |
|
||||||
|
| `inputs.party_file` | string | No | legacy-party override; forbidden for a canonical campaign party |
|
||||||
|
| `inputs.spell_catalog_file` | string | No | overrides the optional campaign spell catalog; empty or omitted inherits the campaign value |
|
||||||
|
| `inputs.audio_dir` | string | Conditional | local audio mode |
|
||||||
|
| `inputs.audio_files[]` | list[string] | Conditional | local audio mode |
|
||||||
|
| `inputs.audio_s3.prefix` | string | Conditional | S3 audio mode |
|
||||||
|
|
||||||
|
Audio rules:
|
||||||
|
|
||||||
|
- configure local mode (`audio_dir` or `audio_files`) or S3 mode (`audio_s3.prefix`), not both.
|
||||||
|
- `audio_s3` requires `pipeline.storage.backend: s3` and a configured S3 bucket.
|
||||||
|
|
||||||
|
### Storage backend selection
|
||||||
|
|
||||||
|
`local` is the default and disables remote object-store operations. Configure
|
||||||
|
`s3` explicitly before supplying `storage.s3`; a populated S3 block does not
|
||||||
|
select a backend on its own. Unknown backend names and an S3 block paired with
|
||||||
|
`local` are rejected during configuration validation.
|
||||||
|
|
||||||
|
### Previous-session expectation
|
||||||
|
|
||||||
|
`previous_session_id` is optional in a session file. When a command supplies
|
||||||
|
`--previous-session-id`, however, the session file must contain the same value;
|
||||||
|
an omitted or different value is rejected before the command performs work.
|
||||||
|
|
||||||
|
## Maintained Examples
|
||||||
|
|
||||||
|
See the [maintained examples index](../examples/README.md) for complete pipeline,
|
||||||
|
campaign, session, template, and input fixtures. Keep complete copyable files
|
||||||
|
there rather than duplicating them in this reference.
|
||||||
|
|||||||
@@ -1,92 +1,59 @@
|
|||||||
# Development Guide
|
# Development
|
||||||
|
|
||||||
## Purpose
|
This is the first-read landing page for people and LLM coding agents working on
|
||||||
Canonical contributor workflow and engineering conventions for implemented Narratio behavior.
|
Narratio. It provides a concise repository orientation and routes each kind of
|
||||||
|
change to its canonical documentation.
|
||||||
|
|
||||||
## Repository layout
|
Narratio is a stage-driven Go orchestrator for turning D&D session audio into
|
||||||
|
polished transcripts and generated artifacts. Start with the
|
||||||
|
[README](../README.md) for product context,
|
||||||
|
[Architecture](policy/architecture.md) for normative system boundaries, and the
|
||||||
|
[Internal Overview](internal/overview.md) for implemented component ownership.
|
||||||
|
|
||||||
- `cmd/narratio/`: CLI entrypoint.
|
## What To Read
|
||||||
- `internal/app/`: command handlers, plan/run/resume orchestration, cleanup gates, secrets loading.
|
|
||||||
- `internal/config/`: strict YAML loading, defaults, and validation.
|
|
||||||
- `internal/stage/`: stage implementations and stage registry/order.
|
|
||||||
- `internal/adapters/`: external boundary adapters (WhisperX, Seriatim, Audita, Scriptorium, storage, notify).
|
|
||||||
- `internal/manifest/`: session/run manifest types and persistence.
|
|
||||||
- `internal/artifacts/`: canonical local/remote path helpers and local artifact store.
|
|
||||||
- `docs/`: canonical documentation set.
|
|
||||||
- `examples/`: maintained config examples used by tests.
|
|
||||||
|
|
||||||
## Build and test commands
|
| When working on | Read | Why |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| Finding the package or component that owns current behavior | [Internal Overview](internal/overview.md) | It is the implemented component inventory and routes to focused internal documents. |
|
||||||
|
| Application shape, boundaries, dependency direction, runtime invariants, safety properties, or dependencies | [Architecture](policy/architecture.md) | It defines the intended system shape, ownership, and non-goals. |
|
||||||
|
| Any documentation addition or revision | [Documentation Policy](policy/documentation.md) | It defines canonical owners, audiences, current-behavior rules, and maintenance requirements. |
|
||||||
|
| Adding, changing, reviewing, rewriting, or deleting tests | [Testing Policy](policy/testing.md) | It defines risk-based sufficiency, durable test boundaries, test-double guidance, and test lifecycle decisions. |
|
||||||
|
| CLI composition or command behavior | [Internal Overview](internal/overview.md) and [CLI Reference](cli.md) | The overview routes to command ownership; the reference owns public syntax and invocation behavior. |
|
||||||
|
| Configuration loading, resolution, or user-visible configuration | [Internal Overview](internal/overview.md) and [Configuration](config.md) | The overview routes to implementation ownership; the reference owns fields, defaults, discovery, and validation. |
|
||||||
|
| Session workflow, status, restore, cleanup, or object storage | [Restore Internals](internal/command-restore.md), [Workspace Internals](internal/workspace.md), [Storage Internals](internal/storage.md), [Operations](operations.md), and [Troubleshooting](troubleshooting.md) | These separate implementation mechanics, operator procedures, and symptom-driven recovery. |
|
||||||
|
| Pipeline sequencing or the behavior of a stage | [Internal Overview](internal/overview.md) and its focused stage documents | The overview owns the implemented stage inventory and routes to each stage contract. |
|
||||||
|
| Adapters or external tool contracts | [Adapter Internals](internal/adapters.md) and [Integration Contracts](integrations/README.md) | The internal guide owns adapter composition and mechanics; integration documents own external formats and protocols. |
|
||||||
|
| Manifests, artifacts, workspace paths, or publish behavior | [Manifest Internals](internal/manifest.md), [Artifact Internals](internal/artifacts.md), [Workspace Internals](internal/workspace.md), [Publish Internals](internal/stage-publish.md), and [Operations](operations.md) | These separate implementation state and resolution from operator-visible layout and lifecycle. |
|
||||||
|
| Maintained configuration or input examples | [Configuration](config.md) and [Examples](../examples/README.md) | The reference owns field meanings; the examples directory owns complete copyable files. |
|
||||||
|
| Preparing, validating, or publishing a release | [Release Procedure](release.md) | The maintainer procedure owns version selection, candidate validation, guarded tag publication, and optional later CI inspection. |
|
||||||
|
| Proposed or unimplemented behavior | `docs/roadmap/` | Future work belongs only in roadmap documentation until implemented. |
|
||||||
|
|
||||||
- Run focused CLI behavior checks:
|
For an existing subsystem, also inspect its focused tests and package-level
|
||||||
|
contracts before changing behavior.
|
||||||
|
|
||||||
```bash
|
## Validation
|
||||||
go test ./internal/app -run TestExecute -v
|
|
||||||
```
|
|
||||||
|
|
||||||
- Run config example load/validate checks:
|
Use focused package tests while iterating. Every pull request and push runs the
|
||||||
|
following repository-wide checks before it can be accepted:
|
||||||
|
|
||||||
```bash
|
```sh
|
||||||
go test ./internal/config -run TestExamplesLoadAndValidate -v
|
|
||||||
```
|
|
||||||
|
|
||||||
- Run full test suite:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
go test ./...
|
go test ./...
|
||||||
|
go test -race ./...
|
||||||
|
go vet ./...
|
||||||
|
go build ./...
|
||||||
|
go test ./internal/doccheck
|
||||||
|
go test ./internal/config -run '^TestExamplesLoadAndValidate$'
|
||||||
```
|
```
|
||||||
|
|
||||||
## Coding conventions
|
The documentation check verifies local Markdown links and the dependency graph
|
||||||
|
of the Woodpecker workflows. The configuration check loads every maintained
|
||||||
|
pipeline and session example. Tag CI reuses this validation path before its
|
||||||
|
asynchronous asset publication; the maintainer release boundary is documented
|
||||||
|
in the [Release Procedure](release.md).
|
||||||
|
|
||||||
- Keep orchestration explicit and stage-driven; do not introduce generic workflow/DAG abstractions.
|
Woodpecker also runs `go test -race -shuffle=on -count=3 ./...` on its scheduled
|
||||||
- Keep external-system details inside adapter packages; stages should consume Narratio-level contracts only.
|
job to expose ordering and repeatability defects. Current runners cross-compile
|
||||||
- Use centralized path helpers from `internal/artifacts` rather than ad hoc path concatenation.
|
for macOS and Windows, but do not provide native macOS or Windows execution.
|
||||||
- Preserve manifest-driven state transitions (`running`, `succeeded`, `failed`, `skipped`, `stale`) as the source of run progress.
|
Those cross-builds establish compilation only, not platform-equivalent runtime
|
||||||
- Keep user/operator docs implementation-accurate; planned work belongs only under `docs/roadmap/`.
|
evidence. Add native checks only when official runner labels and successful
|
||||||
|
native-run evidence are available.
|
||||||
For design principles and invariants, see [docs/architecture.md](./architecture.md). For stage/adapter contracts, see [docs/internal/README.md](./internal/README.md).
|
|
||||||
|
|
||||||
## Dependency policy
|
|
||||||
|
|
||||||
- Prefer Go standard library where practical.
|
|
||||||
- Add third-party dependencies only when they provide clear value for required behavior.
|
|
||||||
- Keep dependency additions narrow to the boundary package that needs them.
|
|
||||||
|
|
||||||
## Change playbooks
|
|
||||||
|
|
||||||
### Add config fields
|
|
||||||
|
|
||||||
1. Add fields to config structs in `internal/config`.
|
|
||||||
2. Set defaults in `internal/config/defaults.go` when appropriate.
|
|
||||||
3. Add validation rules in `internal/config/validate.go`.
|
|
||||||
4. Add or update load/validate tests in `internal/config/*_test.go`.
|
|
||||||
5. Update canonical config docs and examples:
|
|
||||||
- [docs/config.md](./config.md)
|
|
||||||
- relevant files under `examples/`
|
|
||||||
|
|
||||||
### Add CLI flags or commands
|
|
||||||
|
|
||||||
1. Update command parsing and behavior in `internal/app`.
|
|
||||||
2. Add or update command tests (`TestExecute` and command-specific tests).
|
|
||||||
3. Update [docs/cli.md](./cli.md) and, if operator workflow changes, [docs/operations.md](./operations.md).
|
|
||||||
|
|
||||||
### Add or modify stages/adapters
|
|
||||||
|
|
||||||
1. Implement stage behavior in `internal/stage` with clear input/output boundaries.
|
|
||||||
2. Keep external transport/subprocess details in `internal/adapters`.
|
|
||||||
3. Preserve manifest and promotion semantics expected by runner and archive logic.
|
|
||||||
4. Add/update stage and adapter tests.
|
|
||||||
5. Update internal component contracts in `docs/internal/`.
|
|
||||||
|
|
||||||
### Update examples
|
|
||||||
|
|
||||||
1. Keep canonical examples only in `examples/`.
|
|
||||||
2. Ensure examples load and validate through runtime config paths.
|
|
||||||
3. Update `internal/config/load_validate_test.go` as needed.
|
|
||||||
4. Update links in `docs/config.md` if example filenames change.
|
|
||||||
|
|
||||||
### Update docs and roadmap
|
|
||||||
|
|
||||||
1. Keep implemented behavior in canonical docs (`README`, `docs/*.md`, `docs/internal/`).
|
|
||||||
2. Keep planned/unimplemented behavior only in `docs/roadmap/`.
|
|
||||||
3. After completing roadmap items, remove or mark them complete in `docs/roadmap/documentation.md`.
|
|
||||||
4. Run a link/path sweep before finalizing changes.
|
|
||||||
|
|||||||
@@ -1,356 +0,0 @@
|
|||||||
# Go Project Documentation Policy
|
|
||||||
|
|
||||||
## Purpose
|
|
||||||
|
|
||||||
Project documentation must help four audiences:
|
|
||||||
|
|
||||||
1. users who need to run the application;
|
|
||||||
2. administrators/operators who need to configure and operate it;
|
|
||||||
3. developers who need to understand and change it safely;
|
|
||||||
4. LLM coding agents that need clear scope, boundaries, and invariants.
|
|
||||||
|
|
||||||
Docs should be accurate, concise, task-oriented, and organized by audience. Prefer links to canonical docs over repetition.
|
|
||||||
|
|
||||||
## Core Rules
|
|
||||||
|
|
||||||
### 1. Keep docs concise
|
|
||||||
|
|
||||||
Each document should cover a defined scope and only the essentials for that scope.
|
|
||||||
|
|
||||||
Avoid:
|
|
||||||
- long background explanations;
|
|
||||||
- repeated reference material;
|
|
||||||
- implementation detail in user-facing docs;
|
|
||||||
- aspirational language outside roadmap docs;
|
|
||||||
- verbose examples where one minimal example is clearer.
|
|
||||||
|
|
||||||
### 2. Document only implemented behavior outside roadmap files
|
|
||||||
|
|
||||||
Unimplemented, planned, aspirational, experimental, or future work may be described only under:
|
|
||||||
|
|
||||||
- `docs/roadmap/`
|
|
||||||
|
|
||||||
No other documentation file, including `README.md`, should describe code, features, modules, stages, commands, config fields, or behaviors that do not currently exist.
|
|
||||||
|
|
||||||
If a feature is partial, non-roadmap docs may describe only the implemented portion and its current boundary.
|
|
||||||
|
|
||||||
### 3. Use canonical homes
|
|
||||||
|
|
||||||
Each type of information should have one canonical location.
|
|
||||||
|
|
||||||
Canonical homes:
|
|
||||||
|
|
||||||
- project purpose and quickstart: `README.md`
|
|
||||||
- development principles: `docs/architecture.md`
|
|
||||||
- configuration reference: `docs/config.md`
|
|
||||||
- CLI reference: `docs/cli.md`
|
|
||||||
- operations and recovery: `docs/operations.md`
|
|
||||||
- troubleshooting: `docs/troubleshooting.md`
|
|
||||||
- implemented internals: `docs/internal/`
|
|
||||||
- future work: `docs/roadmap/`
|
|
||||||
- contributor workflow: `docs/development.md`
|
|
||||||
- copyable examples: `examples/`
|
|
||||||
|
|
||||||
Other files should summarize briefly and link to the canonical source.
|
|
||||||
|
|
||||||
### 4. Keep examples real
|
|
||||||
|
|
||||||
Examples should be valid, maintained, and free of secrets.
|
|
||||||
|
|
||||||
Where practical:
|
|
||||||
- example configs should load successfully;
|
|
||||||
- example commands should match real CLI syntax;
|
|
||||||
- important examples should be covered by tests.
|
|
||||||
|
|
||||||
## Documentation Profiles
|
|
||||||
|
|
||||||
All projects require:
|
|
||||||
|
|
||||||
- `README.md`
|
|
||||||
- `docs/architecture.md`
|
|
||||||
|
|
||||||
Additional docs depend on the project.
|
|
||||||
|
|
||||||
### Small library
|
|
||||||
|
|
||||||
Recommended:
|
|
||||||
- `docs/development.md`, if contributor conventions are non-obvious
|
|
||||||
|
|
||||||
### Simple CLI
|
|
||||||
|
|
||||||
Required:
|
|
||||||
- `docs/cli.md`
|
|
||||||
|
|
||||||
Recommended:
|
|
||||||
- `docs/development.md`
|
|
||||||
|
|
||||||
### Config-driven CLI
|
|
||||||
|
|
||||||
Required:
|
|
||||||
- `docs/cli.md`
|
|
||||||
- `docs/config.md`
|
|
||||||
|
|
||||||
Recommended:
|
|
||||||
- `examples/`
|
|
||||||
- `docs/development.md`
|
|
||||||
|
|
||||||
### Stateful or operator-facing application
|
|
||||||
|
|
||||||
Required:
|
|
||||||
- `docs/cli.md`, if CLI-based
|
|
||||||
- `docs/config.md`, if config-driven
|
|
||||||
- `docs/operations.md`
|
|
||||||
|
|
||||||
Recommended:
|
|
||||||
- `docs/troubleshooting.md`
|
|
||||||
- `examples/`
|
|
||||||
- `docs/development.md`
|
|
||||||
|
|
||||||
### Modular, staged, service-oriented, or orchestration application
|
|
||||||
|
|
||||||
Required:
|
|
||||||
- `docs/cli.md`, if CLI-based
|
|
||||||
- `docs/config.md`, if config-driven
|
|
||||||
- `docs/operations.md`
|
|
||||||
- `docs/internal/`
|
|
||||||
- `docs/development.md`
|
|
||||||
|
|
||||||
Recommended:
|
|
||||||
- `docs/troubleshooting.md`
|
|
||||||
- validated examples under `examples/`
|
|
||||||
|
|
||||||
## Required Documents
|
|
||||||
|
|
||||||
### README.md
|
|
||||||
|
|
||||||
**Audience:** users, administrators, operators
|
|
||||||
|
|
||||||
The README is the outward-facing project orientation page.
|
|
||||||
|
|
||||||
It should include, in order:
|
|
||||||
|
|
||||||
1. concise description;
|
|
||||||
2. elevator pitch;
|
|
||||||
3. shortest useful command or usage example;
|
|
||||||
4. links to targeted docs.
|
|
||||||
|
|
||||||
The README should be short. It is not a manual.
|
|
||||||
|
|
||||||
The “shortest useful command” means the simplest command that performs the project’s core use case. (It does not mean `app --help`.)
|
|
||||||
|
|
||||||
### docs/architecture.md
|
|
||||||
|
|
||||||
**Audience:** developers, LLM coding agents
|
|
||||||
|
|
||||||
`docs/architecture.md` is required for every project.
|
|
||||||
|
|
||||||
It is an inward-facing development policy document. It should describe how the project is intended to be built and changed.
|
|
||||||
|
|
||||||
It should include:
|
|
||||||
|
|
||||||
- project shape;
|
|
||||||
- core design principles;
|
|
||||||
- package and boundary philosophy;
|
|
||||||
- state/persistence philosophy, if applicable;
|
|
||||||
- external integration philosophy, if applicable;
|
|
||||||
- error-handling and logging principles;
|
|
||||||
- testing expectations;
|
|
||||||
- documentation expectations;
|
|
||||||
- architectural invariants;
|
|
||||||
- explicit non-goals, if useful.
|
|
||||||
|
|
||||||
For small projects, this file may be brief. It may simply state that the project is intentionally narrow, monolithic, and dependency-light.
|
|
||||||
|
|
||||||
### docs/config.md
|
|
||||||
|
|
||||||
**Audience:** administrators, operators, advanced users
|
|
||||||
|
|
||||||
Required for applications with configuration files.
|
|
||||||
|
|
||||||
It should include, in order:
|
|
||||||
|
|
||||||
1. config file locations and discovery precedence;
|
|
||||||
2. minimal working config;
|
|
||||||
3. production-oriented config;
|
|
||||||
4. full configuration reference;
|
|
||||||
5. secrets handling, if applicable;
|
|
||||||
6. links to maintained examples.
|
|
||||||
|
|
||||||
The full configuration reference should be canonical.
|
|
||||||
|
|
||||||
### docs/cli.md
|
|
||||||
|
|
||||||
**Audience:** users, administrators, operators
|
|
||||||
|
|
||||||
Required for CLI applications.
|
|
||||||
|
|
||||||
It should include, in order:
|
|
||||||
|
|
||||||
1. shortest useful command;
|
|
||||||
2. command overview;
|
|
||||||
3. complete flag reference;
|
|
||||||
4. common workflows;
|
|
||||||
5. diagnostic or recovery commands, if applicable.
|
|
||||||
|
|
||||||
Explain when commands are useful, not just their syntax.
|
|
||||||
|
|
||||||
### docs/operations.md
|
|
||||||
|
|
||||||
**Audience:** administrators, operators
|
|
||||||
|
|
||||||
Required for applications that maintain state, support resume behavior, run multiple stages, write durable artifacts, use remote storage, or require recovery procedures.
|
|
||||||
|
|
||||||
It should cover:
|
|
||||||
|
|
||||||
- normal workflow;
|
|
||||||
- filesystem layout;
|
|
||||||
- remote storage layout, if applicable;
|
|
||||||
- logs and manifests;
|
|
||||||
- resume/retry behavior;
|
|
||||||
- cleanup behavior;
|
|
||||||
- archive/backup behavior;
|
|
||||||
- safe recovery procedures;
|
|
||||||
- operational caveats.
|
|
||||||
|
|
||||||
### docs/troubleshooting.md
|
|
||||||
|
|
||||||
**Audience:** administrators, operators
|
|
||||||
|
|
||||||
Recommended once recurring failure modes exist.
|
|
||||||
|
|
||||||
Each entry should include:
|
|
||||||
|
|
||||||
- symptom;
|
|
||||||
- likely cause;
|
|
||||||
- diagnostic command or inspection step;
|
|
||||||
- safe fix;
|
|
||||||
- relevant links.
|
|
||||||
|
|
||||||
### docs/development.md
|
|
||||||
|
|
||||||
**Audience:** developers, LLM coding agents
|
|
||||||
|
|
||||||
Required for projects maintained by humans and LLM coding agents.
|
|
||||||
|
|
||||||
It should include:
|
|
||||||
|
|
||||||
- repository layout;
|
|
||||||
- build/test commands;
|
|
||||||
- coding conventions;
|
|
||||||
- dependency policy;
|
|
||||||
- how to add config fields;
|
|
||||||
- how to add CLI flags;
|
|
||||||
- how to add stages/modules/adapters, if applicable;
|
|
||||||
- how to update examples;
|
|
||||||
- documentation update expectations.
|
|
||||||
|
|
||||||
### docs/internal/
|
|
||||||
|
|
||||||
**Audience:** developers, LLM coding agents
|
|
||||||
|
|
||||||
Required for modular, staged, service-oriented, or orchestration projects.
|
|
||||||
|
|
||||||
This directory describes implemented internal components. It is not the roadmap.
|
|
||||||
|
|
||||||
Use one file per major component where useful.
|
|
||||||
|
|
||||||
Each component doc should include:
|
|
||||||
|
|
||||||
1. purpose;
|
|
||||||
2. inputs and outputs;
|
|
||||||
3. boundaries;
|
|
||||||
4. config fields used;
|
|
||||||
5. external adapters used;
|
|
||||||
6. state or manifest behavior, if applicable;
|
|
||||||
7. skip/resume behavior, if applicable;
|
|
||||||
8. failure behavior;
|
|
||||||
9. tests to inspect before changing;
|
|
||||||
10. architectural invariants.
|
|
||||||
|
|
||||||
### docs/roadmap/
|
|
||||||
|
|
||||||
**Audience:** maintainers, developers, LLM coding agents
|
|
||||||
|
|
||||||
This is the only place for planned, future, aspirational, experimental, or unimplemented work.
|
|
||||||
|
|
||||||
Roadmap docs should clearly distinguish:
|
|
||||||
|
|
||||||
- proposed work;
|
|
||||||
- accepted plans;
|
|
||||||
- deferred ideas;
|
|
||||||
- rejected ideas;
|
|
||||||
- implementation prompts or task breakdowns, if useful.
|
|
||||||
|
|
||||||
Roadmap docs should not be confused with current behavior.
|
|
||||||
|
|
||||||
### docs/integrations/
|
|
||||||
|
|
||||||
**Audience:** developers, LLM coding agents
|
|
||||||
|
|
||||||
Required for projects that depend on external CLIs, APIs, services, protocols, or file formats where the integration contract is important to maintain.
|
|
||||||
|
|
||||||
This directory contains concise, versioned reference notes for external integration contracts. It should document only the parts of the external system that this project actually uses.
|
|
||||||
|
|
||||||
Use one file per integration where useful.
|
|
||||||
|
|
||||||
## Examples Directory
|
|
||||||
|
|
||||||
Projects with non-trivial configuration or workflows should include `examples/`.
|
|
||||||
|
|
||||||
Useful examples include:
|
|
||||||
|
|
||||||
- minimal working config;
|
|
||||||
- production-oriented config;
|
|
||||||
- full annotated config;
|
|
||||||
- local development config;
|
|
||||||
- remote/object-storage config;
|
|
||||||
- minimal session/input file.
|
|
||||||
|
|
||||||
Examples should be valid, maintained, tested when practical, and linked from relevant docs.
|
|
||||||
|
|
||||||
## Security and Privacy
|
|
||||||
|
|
||||||
Docs and examples must not include:
|
|
||||||
|
|
||||||
- real API keys;
|
|
||||||
- tokens;
|
|
||||||
- passwords;
|
|
||||||
- private keys;
|
|
||||||
- private environment dumps;
|
|
||||||
- sensitive user data;
|
|
||||||
- raw private transcripts;
|
|
||||||
- private infrastructure details unless intentionally public.
|
|
||||||
|
|
||||||
Document secret-handling mechanisms, not actual secret values.
|
|
||||||
|
|
||||||
## Maintenance Rules
|
|
||||||
|
|
||||||
When docs change, verify the affected behavior.
|
|
||||||
|
|
||||||
Where practical:
|
|
||||||
|
|
||||||
- load example config files in tests;
|
|
||||||
- test CLI examples or command parser behavior;
|
|
||||||
- validate documented flags against real flags;
|
|
||||||
- remove stale references;
|
|
||||||
- update links after renames;
|
|
||||||
- keep roadmap content out of non-roadmap docs.
|
|
||||||
|
|
||||||
If documentation and code disagree, fix the documentation and/or open a roadmap item; do not leave aspirational behavior in current-behavior docs.
|
|
||||||
|
|
||||||
Documentation is complete only when it matches the current code.
|
|
||||||
|
|
||||||
## Documentation Change Checklist
|
|
||||||
|
|
||||||
Before merging documentation changes, verify:
|
|
||||||
|
|
||||||
- README is concise and orientation-focused.
|
|
||||||
- `docs/architecture.md` describes development principles.
|
|
||||||
- Future work appears only under `docs/roadmap/`.
|
|
||||||
- User-facing docs avoid unnecessary internals.
|
|
||||||
- Developer-facing docs preserve boundaries and invariants.
|
|
||||||
- Config examples match the schema.
|
|
||||||
- CLI examples match real commands and flags.
|
|
||||||
- Defaults appear in the canonical config reference.
|
|
||||||
- No secrets or private data are included.
|
|
||||||
- Links are accurate.
|
|
||||||
@@ -1,15 +1,36 @@
|
|||||||
# Integration Documentation Index
|
# Integrations Index
|
||||||
|
|
||||||
## Audience
|
## Audience
|
||||||
Developers and LLM coding agents changing Narratio's external integration contracts.
|
|
||||||
|
Operators, developers, and coding agents who need to understand Narratio's
|
||||||
|
externally observable integration boundaries.
|
||||||
|
|
||||||
## Scope
|
## Scope
|
||||||
Implemented-only reference notes for the external systems Narratio currently integrates with.
|
|
||||||
|
|
||||||
## Integration Docs
|
`docs/integrations/` is the canonical reference for protocols, invocation and
|
||||||
- `audita.md`: Audita adapter invocation and validation contract.
|
data contracts, logical outputs, and compatibility behavior at external tool
|
||||||
- `seriatim.md`: Seriatim normalize/merge/trim adapter contract.
|
boundaries.
|
||||||
- `scriptorium.md`: Scriptorium run/render adapter contract.
|
|
||||||
|
|
||||||
## Canonical Owner
|
These documents describe what Narratio sends or invokes, what it accepts in
|
||||||
`docs/integrations/` is the canonical home for external integration reference notes per `docs/documentation/policy.md`.
|
return, and how failures are surfaced. Internal composition and stage mechanics
|
||||||
|
belong in [the adapter implementation guide](../internal/adapters.md) and the
|
||||||
|
focused stage documents.
|
||||||
|
|
||||||
|
## Integration Contracts
|
||||||
|
|
||||||
|
- [Audita](./audita.md): transcript polishing (`audita process`).
|
||||||
|
- [Notarius](./notarius.md): complete pipeline execution and safe JSON bundle
|
||||||
|
discovery (`notarius run`).
|
||||||
|
- [Party](./party.md): canonical campaign roster input.
|
||||||
|
- [Seriatim](./seriatim.md): merge, normalize, trim, and render operations.
|
||||||
|
- [Scriptorium](./scriptorium.md): artifact generation and debug rendering
|
||||||
|
(`scriptorium run|render`).
|
||||||
|
- [WhisperX](./whisperx.md): speaker-audio transcription over HTTP.
|
||||||
|
|
||||||
|
## Related Canonical Docs
|
||||||
|
|
||||||
|
- [Configuration](../config.md): operator-facing configuration reference.
|
||||||
|
- [Adapter implementation](../internal/adapters.md): shared adapter boundary and
|
||||||
|
runner wiring.
|
||||||
|
- [Internal documentation](../internal/overview.md): stage-specific integration
|
||||||
|
usage and component ownership.
|
||||||
|
|||||||
@@ -1,66 +1,67 @@
|
|||||||
# Integration: audita
|
# Integration: Audita
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Define Narratio's adapter contract for transcript polishing via Audita CLI subprocess execution.
|
Define the Audita adapter contract used by the `polish` stage.
|
||||||
|
|
||||||
## Inputs and Outputs
|
## External Boundary
|
||||||
Inputs (`audita.PolishRequest`):
|
|
||||||
- merged transcript path
|
|
||||||
- glossary path
|
|
||||||
- output processed transcript path
|
|
||||||
- optional report path (required when report enabled)
|
|
||||||
- work dir
|
|
||||||
- generated config path
|
|
||||||
- stdout/stderr log paths
|
|
||||||
- optional module/model/base URL and concurrency knobs
|
|
||||||
|
|
||||||
Outputs (`audita.PolishResult`):
|
Narratio invokes `audita process` as a subprocess for each polish operation.
|
||||||
- processed transcript path
|
The configured timeout and parent cancellation bound the invocation. Internal
|
||||||
- optional report path
|
runner composition is documented in
|
||||||
- generated config path
|
[the adapter implementation guide](../internal/adapters.md).
|
||||||
- stdout/stderr log paths
|
|
||||||
- exit code, duration, invoked binary
|
|
||||||
- adapter metadata
|
|
||||||
|
|
||||||
## Boundaries
|
## Request Contract
|
||||||
Owns:
|
`PolishRequest` carries:
|
||||||
- Deterministic CLI argument construction for `audita process`
|
- required transcript/glossary/output/work-dir paths;
|
||||||
- Environment bridging for API credentials
|
- optional report path (required when report mode is enabled);
|
||||||
- Invocation config emission
|
- generated config and stdout/stderr log paths;
|
||||||
- Output validation for processed transcript and report
|
- optional per-invocation module override.
|
||||||
|
|
||||||
Does not own:
|
The constructed runner owns static Audita settings: binary, timeout,
|
||||||
- Upstream/downstream stage orchestration
|
credentials, default modules, model and endpoint settings, validation and output
|
||||||
- Credential sourcing policy beyond required env-var presence check
|
settings, report mode, and concurrency. The `polish` stage supplies only
|
||||||
|
invocation-specific paths and may override modules for that invocation.
|
||||||
|
|
||||||
## Config Fields Used
|
## Result Contract
|
||||||
Via `pipeline.audita.*` mapped in app/stage wiring:
|
`PolishResult` returns:
|
||||||
- `binary`, `timeout`, `llm_api_key_env`, `modules`, `base_url`, `model`
|
- processed transcript path;
|
||||||
- `transcript_description`, `config_path`, `output_schema`, `work_dir_retention`
|
- optional report path;
|
||||||
- `total_llm_concurrency`, `proposal_llm_concurrency`, `validation_model`, `validation_llm_concurrency`, `report`
|
- work dir and generated-config/log paths;
|
||||||
|
- exit code, duration, binary provenance;
|
||||||
|
- adapter metadata map.
|
||||||
|
|
||||||
## External Adapters Used
|
## Validation and Failure Semantics
|
||||||
- Shared subprocess helper (`internal/adapters/subprocess`) to run CLI and capture logs.
|
Construction fails for invalid static config values, including:
|
||||||
|
- empty binary;
|
||||||
|
- non-positive timeout;
|
||||||
|
- invalid base URL;
|
||||||
|
- invalid output schema;
|
||||||
|
- invalid work-dir retention value;
|
||||||
|
- invalid concurrency values.
|
||||||
|
|
||||||
## State and Manifest Behavior
|
Run fails for:
|
||||||
- No direct manifest writes.
|
- missing required request paths;
|
||||||
- Stage-level metadata records adapter provenance and credential-present signal.
|
- missing required credential env var when configured (`llm_api_key_env`);
|
||||||
- Generated invocation YAML is written when `GeneratedConfigPath` is provided.
|
- subprocess execution failure;
|
||||||
|
- invalid processed transcript JSON (`segments` array required);
|
||||||
|
- invalid report JSON when reporting is enabled.
|
||||||
|
|
||||||
## Skip and Resume Behavior
|
Processed transcript JSON is limited to 64 MiB and optional report JSON to 16
|
||||||
- Adapter has no skip/resume logic. Stage/runner controls this.
|
MiB. Both must be regular files without symlinked path components.
|
||||||
|
|
||||||
## Failure Behavior
|
Failure results still include output/log/config/exit metadata for diagnostics.
|
||||||
- Constructor validation fails on invalid binary/timeout/schema/concurrency/URL values.
|
|
||||||
- Run fails on missing required paths, missing required credential env var, subprocess errors, invalid processed JSON shape, or invalid report JSON.
|
|
||||||
- Failures preserve stdout/stderr paths in returned result metadata.
|
|
||||||
|
|
||||||
## Tests to Inspect Before Changing
|
## Deterministic Behavior
|
||||||
- `internal/adapters/audita/subprocess_test.go`
|
- CLI args are built from runner config + request in a fixed order.
|
||||||
- `internal/adapters/audita/fake_test.go`
|
- Generated invocation YAML (`audita.generated.v1`) is emitted when requested.
|
||||||
- `internal/stage/polish_test.go`
|
- Manifest writes are stage-owned; adapter itself is stateless.
|
||||||
|
|
||||||
## Architectural Invariants
|
## Configuration
|
||||||
- Processed output must be valid JSON with top-level `segments` array.
|
|
||||||
- When report is enabled, report output must be valid JSON.
|
Operator-selected values are defined under `pipeline.audita.*` in the
|
||||||
- If `llm_api_key_env` is configured, credential must be present in environment.
|
[configuration reference](../config.md#pipeline).
|
||||||
|
|
||||||
|
Maintained example with Audita config:
|
||||||
|
|
||||||
|
- [Full annotated pipeline](../../examples/pipeline.full.annotated.yml)
|
||||||
|
- [Production-shaped pipeline](../../examples/pipeline.production.yml)
|
||||||
|
|||||||
132
docs/integrations/notarius.md
Normal file
132
docs/integrations/notarius.md
Normal file
@@ -0,0 +1,132 @@
|
|||||||
|
# Notarius Integration Contract
|
||||||
|
|
||||||
|
## Boundary
|
||||||
|
|
||||||
|
Narratio uses Notarius as a subprocess to extract configured structured JSON
|
||||||
|
lanes from the final trimmed Seriatim transcript. Narratio owns invocation,
|
||||||
|
safe bundle discovery, lane selection, and its own artifact metadata. Notarius
|
||||||
|
owns pipeline definitions, lane schemas, the receipt, and bundle formats.
|
||||||
|
|
||||||
|
Canonical Notarius v0.6.0 references:
|
||||||
|
|
||||||
|
- [CLI reference](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/cli.md)
|
||||||
|
- [Subprocess consumer contract](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/consumers/subprocess.md)
|
||||||
|
- [D&D pipeline and lane contracts](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/consumers/dnd-pipeline.md)
|
||||||
|
- [Run-result receipt](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/integrations/run-result.md)
|
||||||
|
- [JSON output bundle](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/integrations/json-output.md)
|
||||||
|
- [D&D spell-catalog overlay](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/integrations/dnd-spell-catalog-overlays.md)
|
||||||
|
|
||||||
|
The [complete Narratio example](../../examples/pipeline.full.annotated.yml)
|
||||||
|
records the exact current constraints for all ten D&D lanes. Treat the linked
|
||||||
|
Notarius documents as canonical when changing those values; Narratio does not
|
||||||
|
duplicate the complete schemas.
|
||||||
|
|
||||||
|
## Invocation
|
||||||
|
|
||||||
|
When `pipeline.notarius.enabled` is true, Narratio resolves the executable,
|
||||||
|
configuration path, input path, output directory, and working directory to
|
||||||
|
absolute paths. Narratio requires the Notarius v0.6.0 CLI contract when
|
||||||
|
references are configured and invokes each binding as a separate argument
|
||||||
|
before `--json`:
|
||||||
|
|
||||||
|
```text
|
||||||
|
notarius run <pipeline_id> --config <config_path> --input <trimmed_json> --output-dir <staging_dir> [--reference <selector>=<verified_snapshot_path>]... --json
|
||||||
|
```
|
||||||
|
|
||||||
|
Reference paths are absolute invocation-local snapshots streamed from the
|
||||||
|
manifest-verified canonical files prepared inside the current Narratio session
|
||||||
|
workspace. Narratio verifies snapshot checksum and size before and after the
|
||||||
|
subprocess, and passes only configured bindings, ordered lexically by normalized
|
||||||
|
selector, as direct argument-vector entries without shell interpretation. A CLI
|
||||||
|
binding takes precedence over a matching external path in Notarius
|
||||||
|
configuration. Narratio never emits `--without-reference`.
|
||||||
|
|
||||||
|
For canonical party configuration, the `party` binding is the unchanged,
|
||||||
|
validated authored roster and the `players` binding is its generated
|
||||||
|
projection. Both retain their established `narratio.input.party` and
|
||||||
|
`narratio.input.players` source IDs, and both are resolved from the prepared
|
||||||
|
manifest rather than from campaign configuration at extraction time.
|
||||||
|
|
||||||
|
The maintained D&D boundary binds only the four campaign-owned external slots:
|
||||||
|
|
||||||
|
```text
|
||||||
|
notarius run dnd-session \
|
||||||
|
--config <absolute config path> \
|
||||||
|
--input <absolute trimmed transcript path> \
|
||||||
|
--output-dir <absolute staging directory> \
|
||||||
|
--reference glossary=<absolute verified glossary snapshot> \
|
||||||
|
--reference party=<absolute verified party snapshot> \
|
||||||
|
--reference players=<absolute verified players snapshot> \
|
||||||
|
--reference spell_catalog=<absolute verified spell catalog snapshot> \
|
||||||
|
--json
|
||||||
|
```
|
||||||
|
|
||||||
|
The spell-catalog binding is omitted when the campaign does not maintain that
|
||||||
|
optional overlay. Registry, scene-description, combat-turn, and occurrence
|
||||||
|
handoffs generated during the same Notarius run remain in Notarius pipeline
|
||||||
|
composition and must not be emitted as CLI references. The linked CLI and D&D
|
||||||
|
consumer documents own selector targeting, declared slots, media compatibility,
|
||||||
|
and generated-handoff collision rules.
|
||||||
|
|
||||||
|
Standard output is reserved for the JSON receipt. Standard error is captured
|
||||||
|
separately as diagnostic output. Narratio applies the configured timeout and
|
||||||
|
does not interpret stdout as a receipt unless the subprocess exits successfully.
|
||||||
|
It does not pass a Narratio session ID or run `notarius config validate`
|
||||||
|
automatically; the configured working directory and Narratio's minimal child
|
||||||
|
environment apply to the subprocess.
|
||||||
|
|
||||||
|
## Accepted Result
|
||||||
|
|
||||||
|
Narratio's supported invocation baseline is Notarius v0.6.0. The accepted
|
||||||
|
receipt remains `notarius.run-result.v2`; reference flags do not change the
|
||||||
|
receipt or ten-lane output contract. The receipt
|
||||||
|
must identify the configured pipeline, and its `index_file` must be exactly
|
||||||
|
`index.json` beneath the reported bundle root. The production index must name
|
||||||
|
the management files exactly as `manifest.json`, `rejected.json`,
|
||||||
|
`warnings.json`, and `diagnostics.json`. All receipt, index, and lane paths must
|
||||||
|
stay inside that bundle; symlinks and non-regular lane payloads are rejected.
|
||||||
|
|
||||||
|
Supported receipt and index shapes tolerate unknown fields for forward
|
||||||
|
compatibility, while required identity, validation, count, manifest,
|
||||||
|
rejection, warning, diagnostic, and lane-list fields remain mandatory.
|
||||||
|
Narratio applies bounded reads to the receipt, index, rejection, warning, and
|
||||||
|
diagnostic documents. Warning and diagnostic envelopes, group counts,
|
||||||
|
occurrence counts, truncation state, framework-owned origins, and
|
||||||
|
receipt-to-bundle counts must be internally consistent. Optional chunk-map and
|
||||||
|
evidence-context descriptors must carry their complete generic contract
|
||||||
|
metadata when present.
|
||||||
|
|
||||||
|
For every entry in `pipeline.notarius.outputs`, Narratio requires exactly one
|
||||||
|
index descriptor with the configured lane ID, media type, schema ID, schema
|
||||||
|
version, and, when configured, module key. Missing, duplicate, rejected, or
|
||||||
|
incompatible required lanes fail extraction even if Notarius exited zero. A
|
||||||
|
configured lane whose v2 validation summary is `rejected` or `incomplete` also
|
||||||
|
fails extraction.
|
||||||
|
Unconfigured lanes may remain in the preserved bundle but do not become
|
||||||
|
selectable Narratio sources.
|
||||||
|
|
||||||
|
Each accepted configured lane is registered as
|
||||||
|
`narratio.extraction.<output_key>`. The bundle index is retained for audit and
|
||||||
|
resume validation but is not selectable. Scriptorium and publish rules consume
|
||||||
|
only explicitly named lane sources; `--artifacts` never selects Notarius lanes.
|
||||||
|
|
||||||
|
## Failure And Compatibility Behavior
|
||||||
|
|
||||||
|
- Startup and nonzero-exit errors fail extraction and retain captured diagnostics.
|
||||||
|
- Invalid receipt JSON or an unsupported receipt schema fails before bundle use.
|
||||||
|
- Unsafe or incompatible index data and required-lane rejection fail before the
|
||||||
|
staged bundle is promoted to durable storage.
|
||||||
|
- Contract and external provenance metadata are preserved on lane artifact
|
||||||
|
records and through explicit publication.
|
||||||
|
- Undeclared selectors, incompatible reference files, and external/generated
|
||||||
|
reference collisions are Notarius errors and fail extraction normally.
|
||||||
|
|
||||||
|
Rejection, validation, warning, and diagnostic summaries retain bounded stable
|
||||||
|
identity, category, origin, reason-code, status, and occurrence fields without
|
||||||
|
copying free-form external messages into Narratio manifest metadata or reading
|
||||||
|
lane payload bodies.
|
||||||
|
|
||||||
|
Configuration fields and defaults are in [Configuration](../config.md).
|
||||||
|
Operator paths, rerun procedures, and bundle retention are in
|
||||||
|
[Operations](../operations.md). See [Troubleshooting](../troubleshooting.md)
|
||||||
|
for failure recovery.
|
||||||
73
docs/integrations/party.md
Normal file
73
docs/integrations/party.md
Normal file
@@ -0,0 +1,73 @@
|
|||||||
|
# Canonical Party Input
|
||||||
|
|
||||||
|
`party.yml` is a campaign-owned roster input. Narratio recognizes the
|
||||||
|
versioned `narratio.party.v1` document below when it resolves a pipeline,
|
||||||
|
campaign, and session together.
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
schema_version: narratio.party.v1
|
||||||
|
|
||||||
|
characters:
|
||||||
|
arannis:
|
||||||
|
player:
|
||||||
|
name: Eric
|
||||||
|
character:
|
||||||
|
name: Arannis
|
||||||
|
alias:
|
||||||
|
- Ari
|
||||||
|
- The Grey Owl
|
||||||
|
classes:
|
||||||
|
- name: wizard
|
||||||
|
level: 8
|
||||||
|
```
|
||||||
|
|
||||||
|
`characters` is a non-empty mapping. Each key is a stable character ID using
|
||||||
|
the configured-artifact key grammar: a lowercase ASCII letter followed by zero
|
||||||
|
or more lowercase ASCII letters, digits, or underscores. Character order is
|
||||||
|
preserved where roster order matters.
|
||||||
|
|
||||||
|
Every entry has `player.name`, `character.name`, and a non-empty
|
||||||
|
`character.classes` list. Class entries require a non-empty `name` and may
|
||||||
|
include a positive integer `level`. The optional, intentionally singular
|
||||||
|
`character.alias` field is a list. Names, aliases, and class names must be
|
||||||
|
non-empty, trimmed display strings without control characters. Character names
|
||||||
|
and aliases must be unique across the full roster under Unicode-aware
|
||||||
|
case-insensitive comparison; player names may repeat.
|
||||||
|
|
||||||
|
The document has exactly one YAML document and accepts no unknown fields. A
|
||||||
|
wrong or malformed `schema_version` is an error.
|
||||||
|
|
||||||
|
## Legacy migration boundary
|
||||||
|
|
||||||
|
An unversioned party input remains supported only as opaque legacy reference
|
||||||
|
material while campaigns migrate. It requires a separate `players_file` and
|
||||||
|
retains the existing session override behavior. It cannot be mixed with a
|
||||||
|
canonical party: canonical campaigns must omit `players_file`, and sessions
|
||||||
|
must not override their party or players inputs.
|
||||||
|
|
||||||
|
Use the canonical document for new campaigns. The configuration rules and
|
||||||
|
source-relative path behavior are defined in the [Configuration Reference](../config.md).
|
||||||
|
|
||||||
|
## Derived players document
|
||||||
|
|
||||||
|
During `prepare`, Narratio copies the canonical party source bytes unchanged
|
||||||
|
to `inputs/party.yml` and writes this deterministic players-only projection to
|
||||||
|
`inputs/players.yml`:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
schema_version: narratio.players.v1
|
||||||
|
players:
|
||||||
|
- name: Eric
|
||||||
|
character:
|
||||||
|
id: arannis
|
||||||
|
name: Arannis
|
||||||
|
alias:
|
||||||
|
- Ari
|
||||||
|
- The Grey Owl
|
||||||
|
```
|
||||||
|
|
||||||
|
There is one entry per character, sorted by stable character ID. Repeated
|
||||||
|
player names remain separate entries. The optional `alias` list retains its
|
||||||
|
declared order and is omitted when empty. The projection carries no class
|
||||||
|
data. Its prepared manifest record is marked `derived_from_party`; it is not a
|
||||||
|
separate user-provided `players_file`.
|
||||||
@@ -1,64 +1,71 @@
|
|||||||
# Integration: scriptorium
|
# Integration: Scriptorium
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Define Narratio's adapter contract for Scriptorium artifact generation and render-debug subprocess invocations.
|
Define the Scriptorium adapter contract used by `analyze` and trim-bounds generation in `trim`.
|
||||||
|
|
||||||
## Inputs and Outputs
|
## External Boundary
|
||||||
Inputs:
|
|
||||||
- `RunArtifactRequest`: binary, config path, prompt/profile IDs, input map, vars map, timeout, output path, logs/config paths, optional API env and working dir
|
|
||||||
- `RenderArtifactRequest`: same core fields for render mode
|
|
||||||
|
|
||||||
Outputs (`ArtifactResult`):
|
Narratio invokes Scriptorium as a subprocess in these modes:
|
||||||
- output path
|
|
||||||
- stdout/stderr log paths
|
|
||||||
- generated config path
|
|
||||||
- exit code and duration
|
|
||||||
- command mode (`run` or `render`)
|
|
||||||
- prompt/profile provenance
|
|
||||||
- validation failure signal
|
|
||||||
- adapter metadata
|
|
||||||
|
|
||||||
## Boundaries
|
- `scriptorium run`
|
||||||
Owns:
|
- `scriptorium render`
|
||||||
- Deterministic CLI arg construction for `scriptorium run` and `scriptorium render`
|
|
||||||
- Common request validation
|
|
||||||
- Invocation config emission
|
|
||||||
- Output existence/non-empty checks
|
|
||||||
- Validation-failure mapping for run exit code 2
|
|
||||||
|
|
||||||
Does not own:
|
The request timeout and parent cancellation bound each invocation. Internal
|
||||||
- Artifact selection policy (`analyze` stage)
|
runner composition is documented in
|
||||||
- Bounds semantic validation (`trim` stage)
|
[the adapter implementation guide](../internal/adapters.md).
|
||||||
|
|
||||||
## Config Fields Used
|
## Request Contract
|
||||||
Via `pipeline.scriptorium.*` and stage-level artifact config:
|
Both request types carry:
|
||||||
- `binary`, `config_path`, `timeout`, `render_debug`
|
- binary/config/prompt/profile IDs;
|
||||||
- artifact-level `prompt_id`, `profile_id`, `timeout`, `inputs`, `vars`, `output_path`
|
- input map and vars map;
|
||||||
|
- output path;
|
||||||
|
- timeout;
|
||||||
|
- generated config + stdout/stderr log paths;
|
||||||
|
- optional API-key env var name;
|
||||||
|
- optional working directory.
|
||||||
|
|
||||||
## External Adapters Used
|
## Result Contract
|
||||||
- Shared subprocess helper (`internal/adapters/subprocess`).
|
`ArtifactResult` returns:
|
||||||
|
- output/log/generated-config paths;
|
||||||
|
- exit code and duration;
|
||||||
|
- command mode (`run` or `render`);
|
||||||
|
- prompt/profile provenance;
|
||||||
|
- `ValidationFailed` marker;
|
||||||
|
- metadata map.
|
||||||
|
|
||||||
## State and Manifest Behavior
|
## Validation and Failure Semantics
|
||||||
- No direct manifest writes.
|
Request validation fails for:
|
||||||
- Stage metadata records adapter outputs and command mode.
|
- missing binary, prompt id, or output path;
|
||||||
- Generated invocation YAML is written when requested.
|
- non-positive timeout;
|
||||||
|
- empty input/var names;
|
||||||
|
- empty input path values;
|
||||||
|
- missing required credential env var when `APIKeyEnv` is set.
|
||||||
|
|
||||||
## Skip and Resume Behavior
|
Run behavior:
|
||||||
- Adapter has no skip/resume logic. Stage/runner controls execution.
|
- subprocess errors propagate with context;
|
||||||
|
- `run` exit code `2` is mapped to `ValidationFailed=true`;
|
||||||
|
- successful subprocess still fails if output file is missing or empty.
|
||||||
|
|
||||||
## Failure Behavior
|
Each artifact result is limited to 64 MiB and must be a regular file without
|
||||||
- Request validation fails for missing binary/prompt/output, invalid timeout, invalid input/var names, or missing required API env var.
|
symlinked path components.
|
||||||
- Subprocess errors bubble with command context.
|
|
||||||
- `run` exit code 2 is treated as `ValidationFailed=true` and surfaced as error by calling stage.
|
|
||||||
- Successful subprocess still fails if output file is missing/empty.
|
|
||||||
|
|
||||||
## Tests to Inspect Before Changing
|
Render behavior:
|
||||||
- `internal/adapters/scriptorium/subprocess_test.go`
|
- subprocess errors propagate;
|
||||||
- `internal/adapters/scriptorium/fake_test.go`
|
- output file must exist and be non-empty.
|
||||||
- `internal/stage/analyze_test.go`
|
|
||||||
- `internal/stage/trim_test.go`
|
|
||||||
|
|
||||||
## Architectural Invariants
|
## Deterministic Behavior
|
||||||
- Both modes require explicit timeout > 0.
|
- input and var maps are sorted into deterministic `--input` and `--var` CLI args.
|
||||||
- Input/var maps are sorted into deterministic CLI argument order.
|
- stage wiring adds `session_id=narratio-session-<session_id>` to every Scriptorium request for sticky upstream routing, overriding any configured `vars.session_id`.
|
||||||
- Run-mode validation failures are represented explicitly, not silently skipped.
|
- generated invocation YAML (`scriptorium.generated.v1`) is emitted when requested.
|
||||||
|
- adapter is stateless and does not own artifact-selection policy.
|
||||||
|
|
||||||
|
## Configuration
|
||||||
|
|
||||||
|
Operator-selected values are defined under `pipeline.scriptorium.*`, including
|
||||||
|
per-artifact settings under `pipeline.scriptorium.artifacts.*`, in the
|
||||||
|
[configuration reference](../config.md#pipeline).
|
||||||
|
|
||||||
|
Maintained examples with Scriptorium config:
|
||||||
|
|
||||||
|
- [Full annotated pipeline](../../examples/pipeline.full.annotated.yml)
|
||||||
|
- [Production-shaped pipeline](../../examples/pipeline.production.yml)
|
||||||
|
|||||||
@@ -1,60 +1,62 @@
|
|||||||
# Integration: seriatim
|
# Integration: Seriatim
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Define Narratio's adapter contract for merge, normalize, and trim subprocess invocations of Seriatim.
|
Define the Seriatim adapter contract used by `merge`, `normalize`, `trim`, and `render`.
|
||||||
|
|
||||||
## Inputs and Outputs
|
## External Boundary
|
||||||
Inputs:
|
|
||||||
- `MergeRequest`: raw/normalized transcript inputs, output path, optional report, speaker/autocorrect paths, logs/config
|
|
||||||
- `NormalizeRequest`: input transcript, output path, schema, optional report, timeout/log/config
|
|
||||||
- `TrimRequest`: input transcript, output path, keep selector, timeout/log/config
|
|
||||||
|
|
||||||
Outputs:
|
Narratio invokes Seriatim as a subprocess in these modes:
|
||||||
- `MergeResult`, `NormalizeResult`, `TrimResult` with output paths, logs/config paths, exit code, duration, binary provenance, and metadata.
|
|
||||||
|
|
||||||
## Boundaries
|
- `seriatim merge`
|
||||||
Owns:
|
- `seriatim normalize`
|
||||||
- Validated deterministic CLI invocation construction
|
- `seriatim trim`
|
||||||
- Optional env tuning propagation for merge
|
- `seriatim render`
|
||||||
- Invocation config file emission
|
|
||||||
- JSON output validation
|
|
||||||
|
|
||||||
Does not own:
|
The configured timeout and parent cancellation bound each invocation. Internal
|
||||||
- Transcript input selection/promotion logic (stage-owned)
|
runner composition is documented in
|
||||||
- Bounds computation (scriptorium/trim-stage-owned)
|
[the adapter implementation guide](../internal/adapters.md).
|
||||||
|
|
||||||
## Config Fields Used
|
## Request/Result Contracts
|
||||||
Via `pipeline.seriatim.*` mapped in app/stage wiring:
|
- `MergeRequest`/`MergeResult`: multi-input merge to base transcript, optional report.
|
||||||
- `binary`, `timeout`, `output_schema`, `coalesce_gap`, `report`
|
- `NormalizeRequest`/`NormalizeResult`: transcript normalization with explicit schema.
|
||||||
- `env.overlap_word_run_gap`
|
- `TrimRequest`/`TrimResult`: transcript trimming with required keep selector.
|
||||||
- `env.overlap_word_run_reorder_window`
|
- `RenderRequest`/`RenderResult`: transcript-to-markdown rendering with explicit format and render booleans.
|
||||||
- `env.backchannel_max_duration`
|
|
||||||
- `env.filler_max_duration`
|
|
||||||
|
|
||||||
## External Adapters Used
|
Results include output/log/config paths, timing, exit code, and metadata.
|
||||||
- Shared subprocess helper (`internal/adapters/subprocess`).
|
|
||||||
|
|
||||||
## State and Manifest Behavior
|
## Validation and Failure Semantics
|
||||||
- No direct manifest writes.
|
Runner construction validates:
|
||||||
- Stage metadata consumes adapter result fields and preserves generated config/log references.
|
- binary presence;
|
||||||
|
- timeout > 0;
|
||||||
|
- supported output schema (`seriatim-minimal|seriatim-intermediate|seriatim-full`);
|
||||||
|
- non-negative coalesce gap.
|
||||||
|
|
||||||
## Skip and Resume Behavior
|
Invocation fails on:
|
||||||
- Adapter has no skip/resume logic. Runner controls stage execution.
|
- missing required request paths/inputs;
|
||||||
|
- invalid normalize schema override;
|
||||||
|
- unsupported render format;
|
||||||
|
- subprocess failure;
|
||||||
|
- invalid JSON outputs for merge/normalize/trim;
|
||||||
|
- missing `segments` array for normalize/trim transcript outputs;
|
||||||
|
- empty render output files.
|
||||||
|
|
||||||
## Failure Behavior
|
When report paths are provided/enabled, report files must parse as JSON.
|
||||||
- Constructor fails for invalid binary/timeout/output-schema/coalesce-gap.
|
Each Seriatim JSON or rendered-text result is limited to 64 MiB and must be a
|
||||||
- Merge fails on missing output path/inputs/report path (if enabled), subprocess errors, invalid merged output JSON, invalid report JSON.
|
regular file without symlinked path components.
|
||||||
- Normalize fails on missing input/output, invalid schema, subprocess errors, invalid normalized output JSON shape, invalid report JSON.
|
|
||||||
- Trim fails on missing input/output/keep selector, subprocess errors, invalid trimmed output JSON shape.
|
|
||||||
|
|
||||||
## Tests to Inspect Before Changing
|
## Deterministic Behavior
|
||||||
- `internal/adapters/seriatim/subprocess_test.go`
|
- argument ordering is deterministic per command construction.
|
||||||
- `internal/adapters/seriatim/fake_test.go`
|
- merge env overrides are explicit (`SERIATIM_*`) and only emitted when configured.
|
||||||
- `internal/stage/merge_test.go`
|
- generated invocation YAML (`seriatim.generated.v1`) is emitted when requested.
|
||||||
- `internal/stage/normalize_test.go`
|
- adapter does not write manifests or choose stage inputs.
|
||||||
- `internal/stage/trim_test.go`
|
|
||||||
|
|
||||||
## Architectural Invariants
|
## Configuration
|
||||||
- Supported output schemas are limited to `seriatim-minimal`, `seriatim-intermediate`, `seriatim-full`.
|
|
||||||
- Normalize/trim outputs must include `segments` arrays.
|
Operator-selected values are defined under `pipeline.seriatim.*` and
|
||||||
- Merge/normalize/trim all route through deterministic subprocess invocation.
|
`pipeline.render.*` in the
|
||||||
|
[configuration reference](../config.md#pipeline).
|
||||||
|
|
||||||
|
Maintained examples with Seriatim config:
|
||||||
|
|
||||||
|
- [Full annotated pipeline](../../examples/pipeline.full.annotated.yml)
|
||||||
|
- [Production-shaped pipeline](../../examples/pipeline.production.yml)
|
||||||
|
|||||||
71
docs/integrations/whisperx.md
Normal file
71
docs/integrations/whisperx.md
Normal file
@@ -0,0 +1,71 @@
|
|||||||
|
# Integration: WhisperX
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
|
||||||
|
WhisperX transcribes each prepared speaker audio file for Narratio's
|
||||||
|
`transcribe` stage. Narratio uses an HTTP boundary and installs each successful
|
||||||
|
response as that speaker's raw transcript JSON.
|
||||||
|
|
||||||
|
## HTTP Boundary
|
||||||
|
|
||||||
|
Narratio sends an HTTP `POST` to the configured transcription URL using
|
||||||
|
`multipart/form-data` with:
|
||||||
|
|
||||||
|
- `file`: the audio file, retaining its base filename; and
|
||||||
|
- `language`: the configured language string.
|
||||||
|
|
||||||
|
The server must return a `2xx` response whose body is valid JSON. Narratio does
|
||||||
|
not currently require a more specific response schema at this boundary.
|
||||||
|
|
||||||
|
The transcription URL must be an absolute `http` or `https` URL. The audio body
|
||||||
|
is streamed through a fresh multipart writer for every attempt, so its memory
|
||||||
|
use is bounded by the transport buffer rather than by the complete audio file.
|
||||||
|
WhisperX response acquisition is capped at 10 MiB.
|
||||||
|
|
||||||
|
## Request And Result Contract
|
||||||
|
|
||||||
|
Each adapter request identifies a speaker, a readable audio file, and the
|
||||||
|
destination for the raw transcript. The HTTP request carries the audio and
|
||||||
|
language; the speaker identifier remains Narratio orchestration metadata.
|
||||||
|
|
||||||
|
On success, Narratio atomically writes the response body to the requested
|
||||||
|
destination. The adapter result reports that logical output together with the
|
||||||
|
attempt count, final HTTP status when available, elapsed duration, and adapter
|
||||||
|
identity metadata. A failed or invalid response is not installed as the
|
||||||
|
transcript output.
|
||||||
|
|
||||||
|
## Retry, Timeout, And Cancellation
|
||||||
|
|
||||||
|
- The configured timeout applies independently to each HTTP attempt.
|
||||||
|
- `retries` means additional attempts after the first.
|
||||||
|
- HTTP `429`, HTTP `5xx`, attempt timeouts, and network errors are retryable.
|
||||||
|
- Other HTTP `4xx` responses and explicit cancellation are not retryable.
|
||||||
|
- Narratio waits the configured retry delay between attempts and aborts that
|
||||||
|
wait when the parent context is canceled.
|
||||||
|
|
||||||
|
## Validation And Failure Semantics
|
||||||
|
|
||||||
|
Client construction rejects a missing or non-HTTP(S) absolute transcription URL,
|
||||||
|
a missing language, a non-positive timeout, negative retries, or a negative
|
||||||
|
retry delay. A request fails before transmission when its audio or output path
|
||||||
|
is missing.
|
||||||
|
|
||||||
|
Non-`2xx` status, transport failure, response-size overflow, invalid JSON, or
|
||||||
|
failure to install the output causes the transcription to fail. Errors include
|
||||||
|
attempt context, and the result retains attempts, final status when available,
|
||||||
|
and elapsed duration for diagnostics.
|
||||||
|
|
||||||
|
## Determinism And Concurrency
|
||||||
|
|
||||||
|
Each audio request has stable multipart field names, and successful bytes are
|
||||||
|
installed atomically. The transcribe stage may process speaker files in
|
||||||
|
parallel, bounded by the configured concurrency. It records results in stable
|
||||||
|
speaker order after all work completes; any speaker failure fails the stage.
|
||||||
|
|
||||||
|
## Related Canonical Docs
|
||||||
|
|
||||||
|
- [Configuration](../config.md#pipeline) defines the operator-selected
|
||||||
|
WhisperX URL, language, timeouts, retry policy, and concurrency.
|
||||||
|
- [Adapter implementation](../internal/adapters.md) describes internal wiring.
|
||||||
|
- [Transcribe stage](../internal/stage-transcribe.md) describes stage mechanics,
|
||||||
|
durable artifacts, and manifests.
|
||||||
@@ -1,28 +0,0 @@
|
|||||||
# Internal Documentation Index
|
|
||||||
|
|
||||||
## Audience
|
|
||||||
Developers and LLM coding agents changing Narratio internals.
|
|
||||||
|
|
||||||
## Scope
|
|
||||||
Implementation-accurate contracts for workspace/state, manifests, stages, artifact resolution, and adapter boundaries.
|
|
||||||
|
|
||||||
## Component Docs
|
|
||||||
- `adapters.md`: external adapter map, runtime wiring, and boundary ownership.
|
|
||||||
- `storage.md`: remote storage backend contracts and object-store invariants.
|
|
||||||
- `manifest.md`: session/run manifest schemas, lifecycle transitions, and persistence semantics.
|
|
||||||
- `artifacts.md`: built-in artifact registry, runtime artifact catalog, and source-resolution behavior.
|
|
||||||
- `workspace.md`: local state model, manifests, run-local layout, promotion, and cleanup invariants.
|
|
||||||
- `stage-prepare.md`: input materialization and provenance capture.
|
|
||||||
- `stage-transcribe.md`: WhisperX transcript generation.
|
|
||||||
- `stage-merge.md`: Seriatim normalization + merge.
|
|
||||||
- `stage-polish.md`: Audita transcript polishing.
|
|
||||||
- `stage-normalize.md`: post-polish normalization.
|
|
||||||
- `stage-trim.md`: bounds-driven transcript trimming.
|
|
||||||
- `stage-analyze.md`: dependency-ordered Scriptorium artifact generation for selected configured artifacts.
|
|
||||||
- `stage-archive.md`: archive upload and current-pointer publish contract.
|
|
||||||
|
|
||||||
## External Integration Notes
|
|
||||||
- `../integrations/README.md`: canonical location for external integration contracts (`audita.md`, `seriatim.md`, `scriptorium.md`).
|
|
||||||
|
|
||||||
## Canonical Owner
|
|
||||||
`docs/internal/` is the canonical home for implemented internals per `docs/documentation/policy.md`.
|
|
||||||
@@ -1,79 +1,105 @@
|
|||||||
# Internal: Adapters
|
# Internal: Adapters
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Describe the external adapter boundaries used by Narratio stages and app orchestration, including default runtime wiring.
|
|
||||||
|
|
||||||
## Inputs and outputs
|
Explain the adapter interfaces and production composition used by application
|
||||||
Inputs:
|
and stage orchestration. Externally observable protocols and formats belong in
|
||||||
- Stage requests passed through adapter interfaces (for example transcription, merge/normalize/trim, polish, artifact generation, object-store operations, notifications).
|
the [integration contracts](../integrations/).
|
||||||
- Resolved config values used to construct default adapters.
|
|
||||||
|
|
||||||
Outputs:
|
## Adapter Boundaries
|
||||||
- Adapter-specific result structs (paths, metadata, status/attempt info, duration/exit details).
|
|
||||||
- Adapter errors returned to stage/app orchestration.
|
|
||||||
|
|
||||||
## Boundaries
|
Narratio stage logic depends on adapter interfaces, not transport-specific details.
|
||||||
Owns:
|
|
||||||
- Transport/process/SDK details at system boundaries (`HTTP`, subprocess CLI invocation, AWS SDK calls).
|
|
||||||
- Request/response contracts in `internal/adapters/*` packages.
|
|
||||||
|
|
||||||
Does not own:
|
Primary adapters:
|
||||||
- Stage sequencing, skip/force/resume decisions.
|
|
||||||
- Manifest transition logic.
|
|
||||||
- Canonical workspace path policy.
|
|
||||||
|
|
||||||
## Config fields used
|
|
||||||
Default wiring and adapter calls consume:
|
|
||||||
- `pipeline.whisperx.*`
|
|
||||||
- `pipeline.seriatim.*`
|
|
||||||
- `pipeline.audita.*`
|
|
||||||
- `pipeline.scriptorium.*`
|
|
||||||
- `pipeline.storage.*` and `pipeline.archive.*` (object-store construction/gating)
|
|
||||||
- `pipeline.notification.*` (sender boundary exists; placeholder behavior today)
|
|
||||||
|
|
||||||
## External adapters used
|
|
||||||
Runtime env boundary fields (`internal/stage.Env`):
|
|
||||||
- `whisperx.Client`
|
- `whisperx.Client`
|
||||||
- `seriatim.Runner`
|
- `seriatim.Runner`
|
||||||
- `audita.Runner`
|
- `audita.Runner`
|
||||||
- `scriptorium.Runner`
|
- `scriptorium.Runner`
|
||||||
|
- `notarius.Runner`
|
||||||
- `storage.ObjectStore`
|
- `storage.ObjectStore`
|
||||||
- `notify.Sender`
|
- `notify.Sender`
|
||||||
- `analyzer.Runner`
|
|
||||||
|
|
||||||
Current execution usage:
|
## Ownership
|
||||||
- Actively used by implemented stages: `WhisperX`, `Seriatim`, `Audita`, `Scriptorium`, `ObjectStore`, `Notifier`.
|
|
||||||
- Present but not used by implemented stage set: `Analyzer`, legacy `storage.Backend`.
|
|
||||||
|
|
||||||
Default construction in app runner:
|
Adapters own:
|
||||||
- Auto-constructed when not injected: WhisperX HTTP client, Seriatim subprocess runner, Audita subprocess runner, Scriptorium subprocess runner, object store (only when needed), and `notify.NoopSender`.
|
|
||||||
- Callers can inject test/fake implementations through `app.RunOptions.Env`.
|
|
||||||
|
|
||||||
## State and manifest behavior
|
- HTTP/subprocess/SDK argument and transport details.
|
||||||
- Adapters do not directly mutate session/run manifests.
|
- Backend-specific request/response mapping.
|
||||||
- Stages and runner own manifest writes and stage status transitions.
|
|
||||||
- Adapter outputs are persisted indirectly through stage result mapping (outputs/logs/generated configs/metadata).
|
|
||||||
|
|
||||||
## Skip and resume behavior
|
Adapters do not own:
|
||||||
- No adapter-level skip/resume semantics.
|
|
||||||
- Skip/resume/force behavior is decided by app runner using manifest stage state.
|
|
||||||
|
|
||||||
## Failure behavior
|
- stage ordering/skip/force logic;
|
||||||
- Adapter constructors validate config-derived values and fail early on invalid required inputs.
|
- manifest transitions;
|
||||||
- Adapter run-time failures are returned to stage code with boundary context and are recorded as stage failures by runner logic.
|
- canonical path policy.
|
||||||
- Subprocess adapters preserve stdout/stderr and generated-config paths to aid diagnosis.
|
|
||||||
|
|
||||||
## Tests to inspect before changing
|
## Default Wiring
|
||||||
|
|
||||||
|
`internal/app/runner.go` initializes default adapters when not injected and
|
||||||
|
only when the selected execution plan needs them:
|
||||||
|
|
||||||
|
- WhisperX HTTP client for `transcribe`.
|
||||||
|
- Seriatim subprocess runner for `merge`, `normalize`, `trim`, or `render`.
|
||||||
|
- Audita subprocess runner for `polish`.
|
||||||
|
- Scriptorium subprocess runner for `trim` or `analyze`.
|
||||||
|
- Notarius subprocess runner for `extract` when extraction is enabled.
|
||||||
|
- Noop notifier (`notify.NoopSender`) for `notify`.
|
||||||
|
- Object store only when required by selected stages/config.
|
||||||
|
|
||||||
|
Remote publish locks are loaded only for a selected, enabled publish that
|
||||||
|
uploads a run. Shared session lifecycle setup still applies to every selected
|
||||||
|
range, but an unselected integration is neither initialized nor validated by
|
||||||
|
runner composition. Each selected stage retains its own fail-fast configuration
|
||||||
|
and input validation.
|
||||||
|
|
||||||
|
`session plan` is outside production adapter composition. It performs
|
||||||
|
resume validation and models selected transitions against cloned manifest
|
||||||
|
state without constructing or invoking stage-execution adapters. The shared
|
||||||
|
command configuration loader may still use object storage to retrieve a missing
|
||||||
|
remote session file before planning begins.
|
||||||
|
|
||||||
|
Notarius is composed only when extraction is enabled; the extract stage owns
|
||||||
|
prepared reference resolution, receipt, bundle, and configured-lane policy.
|
||||||
|
The adapter validates the ordered selector/absolute-path pairs and is the sole
|
||||||
|
owner of serializing them as repeated `--reference` arguments before `--json`.
|
||||||
|
|
||||||
|
Object-store construction goes through `newCommandObjectStore`, which loads
|
||||||
|
configured filesystem secrets before adapter initialization.
|
||||||
|
|
||||||
|
## Failure Semantics
|
||||||
|
|
||||||
|
- Constructor errors fail stage execution setup early.
|
||||||
|
- Runtime adapter errors propagate to stage code and then manifest failure handling.
|
||||||
|
- Subprocess adapters persist stage logs/generated configs through stage-managed paths.
|
||||||
|
- Shared subprocess execution starts an owned process group on Linux/macOS or a
|
||||||
|
kill-on-close job object on Windows. Every terminal path disposes of that
|
||||||
|
owned tree before returning. After a natural leader exit, Unix checks for
|
||||||
|
remaining group members and uses bounded graceful then forceful termination;
|
||||||
|
Windows closes the job so kill-on-close applies. Cancellation, deadlines, and
|
||||||
|
diagnostic limits use the same terminal disposal path without losing their
|
||||||
|
original result classification. Child environments contain only the execution
|
||||||
|
baseline and adapter-specified values; configured credentials are explicit
|
||||||
|
sensitive values. Stdout and stderr are redacted while streaming into separate
|
||||||
|
8 MiB diagnostic captures; a bounded wait closes a stream retained by a
|
||||||
|
departed leader's descendant. Unsupported platforms reject owned command
|
||||||
|
execution.
|
||||||
|
|
||||||
|
## Implementation And Tests
|
||||||
|
|
||||||
|
- Composition: `internal/app/runner.go`, `internal/app/object_store.go`
|
||||||
|
- Shared subprocess mechanics: `internal/adapters/subprocess`
|
||||||
|
- Focused adapters: `internal/adapters/{whisperx,seriatim,audita,scriptorium,notarius,storage,notify}`
|
||||||
- `internal/adapters/whisperx/http_test.go`
|
- `internal/adapters/whisperx/http_test.go`
|
||||||
- `internal/adapters/seriatim/subprocess_test.go`
|
- `internal/adapters/seriatim/subprocess_test.go`
|
||||||
- `internal/adapters/audita/subprocess_test.go`
|
- `internal/adapters/audita/subprocess_test.go`
|
||||||
- `internal/adapters/scriptorium/subprocess_test.go`
|
- `internal/adapters/scriptorium/subprocess_test.go`
|
||||||
|
- `internal/adapters/notarius/subprocess_test.go`
|
||||||
- `internal/adapters/storage/*_test.go`
|
- `internal/adapters/storage/*_test.go`
|
||||||
- `internal/adapters/notify/fake_test.go`
|
|
||||||
- `internal/adapters/analyzer/fake_test.go`
|
|
||||||
- `internal/app/runner_test.go`
|
- `internal/app/runner_test.go`
|
||||||
|
|
||||||
## Architectural invariants
|
See the [WhisperX](../integrations/whisperx.md),
|
||||||
- Stage code depends on adapter interfaces, not transport-specific implementation types.
|
[Seriatim](../integrations/seriatim.md), [Audita](../integrations/audita.md),
|
||||||
- External SDK-specific types remain inside adapter implementations.
|
[Scriptorium](../integrations/scriptorium.md), and
|
||||||
- Default app wiring must remain deterministic and overrideable via injected env dependencies.
|
[Notarius](../integrations/notarius.md) contracts before changing an
|
||||||
|
externally visible boundary. Operator-selected values belong in
|
||||||
|
[Configuration](../config.md).
|
||||||
|
|||||||
@@ -1,87 +1,264 @@
|
|||||||
# Internal: Artifacts
|
# Internal: Artifacts
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Define Narratio's artifact identity and resolution model for built-in transcript/bounds artifacts and runtime-configured analyze artifacts.
|
|
||||||
|
|
||||||
## Inputs and outputs
|
Explain the artifact registry, runtime catalog, resolver, previous-input
|
||||||
Inputs:
|
requirements, and shared remote current-state mechanics implemented by
|
||||||
- artifact sources from config/runtime (`pipeline.scriptorium.artifacts.*.inputs.*.source`)
|
`internal/artifacts`. Configuration fields that accept source IDs belong in
|
||||||
- session paths and optional session manifest stage outputs
|
[Configuration](../config.md); physical placement belongs in
|
||||||
- runtime artifact catalog state for configured artifact sources
|
[Operations](../operations.md).
|
||||||
|
|
||||||
Outputs:
|
## Built-in Source IDs
|
||||||
- resolved local artifact path and provenance (`ResolvedSessionArtifact`)
|
|
||||||
- runtime catalog entries for planned/executable/available artifacts
|
|
||||||
- validation errors for unsupported, missing, or invalid artifact sources
|
|
||||||
|
|
||||||
## Boundaries
|
The internal registry recognizes these stable built-in source IDs:
|
||||||
Owns:
|
|
||||||
- built-in artifact registry and content validation rules
|
|
||||||
- runtime artifact catalog for configured artifact source IDs
|
|
||||||
- source resolution behavior for built-in and configured artifact sources
|
|
||||||
|
|
||||||
Does not own:
|
- `narratio.transcript.base`
|
||||||
- artifact generation (stages produce files)
|
- `narratio.transcript.polished`
|
||||||
- manifest transition policy
|
- `narratio.transcript.final`
|
||||||
- archive promotion behavior
|
- `narratio.transcript.final_trimmed`
|
||||||
|
- `narratio.transcript.final_markdown`
|
||||||
|
- `narratio.transcript.final_trimmed_markdown`
|
||||||
|
- `narratio.bounds.session`
|
||||||
|
|
||||||
## Config fields used
|
Registry entries bind each ID to its producer, output kind, canonical fallback,
|
||||||
- `pipeline.scriptorium.artifacts.<name>.enabled`
|
and content validator. The focused stage documents own their input/output flow;
|
||||||
- `pipeline.scriptorium.artifacts.<name>.output_path`
|
[Configuration](../config.md) owns where operators may select these IDs.
|
||||||
- `pipeline.scriptorium.artifacts.<name>.inputs.<key>.source`
|
|
||||||
|
|
||||||
## External adapters used
|
## Configured, Extraction, And Previous-Session Sources
|
||||||
- none
|
|
||||||
|
|
||||||
## State and manifest behavior
|
- configured source ID format: `narratio.artifact.<artifact_key>`
|
||||||
Built-in registry entries:
|
- extraction source ID format: `narratio.extraction.<output_key>`
|
||||||
|
- previous-session source ID format: `narratio.previous_session.artifact.<artifact_key>`
|
||||||
|
|
||||||
| Artifact ID | Canonical file | Producer stage | Output kind |
|
All formats are validated by strict source-policy rules. Configured artifact and
|
||||||
| --- | --- | --- | --- |
|
extraction keys use `^[a-z][a-z0-9_]*$`; source parsers never normalize an
|
||||||
| `narratio.transcript.merged` | `transcripts/merged.json` | `merge` | `transcript_merged` |
|
unrecognized token into a valid source. Extraction sources are registered only
|
||||||
| `narratio.transcript.polished` | `transcripts/processed.json` | `polish` | `transcript_processed` |
|
from `pipeline.notarius.outputs`; the Notarius index has no selectable source
|
||||||
| `narratio.transcript.full` | `transcripts/normalized.json` | `normalize` | `transcript_normalized` |
|
ID.
|
||||||
| `narratio.transcript.trimmed` | `transcripts/trimmed.json` | `trim` | `transcript_trimmed` |
|
|
||||||
| `narratio.bounds.session` | `artifacts/session_bounds.json` | `trim` | `session_bounds` |
|
|
||||||
|
|
||||||
Runtime catalog entries include built-ins and configured `narratio.artifact.<name>` sources.
|
Prepared stable source IDs are `narratio.input.players`,
|
||||||
|
`narratio.input.party`, `narratio.input.glossary`, and
|
||||||
|
`narratio.input.spell_catalog`. Artifact policy owns their canonical manifest
|
||||||
|
kind and prepared filename vocabulary. Canonical party mode preserves the
|
||||||
|
party source bytes in the party record and supplies the players record from the
|
||||||
|
deterministic `derived_from_party` projection; both remain ordinary prepared
|
||||||
|
source IDs for consumers.
|
||||||
|
|
||||||
Catalog states:
|
## Runtime Catalog
|
||||||
- `planned`: source is registered and known for this run
|
|
||||||
- `executable`: configured artifact is selected for analyze execution
|
|
||||||
- `available`: artifact has a usable file path (generated this run or reused from disk)
|
|
||||||
|
|
||||||
Resolution behavior:
|
`ArtifactCatalog` tracks:
|
||||||
- built-in sources resolve via manifest producer outputs first, then canonical fallback path
|
|
||||||
- configured `narratio.artifact.<name>` sources resolve through runtime catalog availability
|
- `planned`: source registered for run context;
|
||||||
- configured source lookup requires catalog context
|
- `executable`: included in the effective analyze artifact set;
|
||||||
|
- `available`: the source's canonical evidence owner validates its current
|
||||||
|
manifest record and durable bytes;
|
||||||
|
- `provenance`: availability source.
|
||||||
|
|
||||||
|
Configured definitions are always registered. Without an explicit selection,
|
||||||
|
the effective analyze set contains enabled definitions. With `--artifacts`, the
|
||||||
|
exact named configured definitions become the effective set for that invocation,
|
||||||
|
regardless of their `enabled` value. The effective-set resolver itself does not
|
||||||
|
expand dependencies; the analyze work planner closes those targets over their
|
||||||
|
configured prerequisite graph. Availability is separate from executability.
|
||||||
|
Configuration may normalize a family selection into its concrete generated
|
||||||
|
members before this resolver runs. The effective set retains optional family
|
||||||
|
and character origin metadata, but its keys, catalog sources, and runtime
|
||||||
|
lookups remain concrete configured-artifact identities.
|
||||||
|
Family publish policies are likewise expanded into ordinary configured-source
|
||||||
|
publish rules during configuration resolution.
|
||||||
|
Configured outputs, including non-executable prerequisites, become available
|
||||||
|
only when the versioned analyze state identifies a current result whose source,
|
||||||
|
contract, canonical configured path, size, and checksum match a confined
|
||||||
|
no-follow regular file. An incidental canonical file and a legacy aggregate
|
||||||
|
analyze output are unavailable.
|
||||||
|
Extraction entries are registered from configuration and become available only
|
||||||
|
after compatible extraction evidence is hydrated.
|
||||||
|
|
||||||
|
During an analyze invocation, a newly validated and atomically materialized
|
||||||
|
configured output is marked available with its producer run ID, contract,
|
||||||
|
checksum, and size. Later scheduled dependents therefore observe the same
|
||||||
|
semantic identity whether their prerequisite was reused from current manifest
|
||||||
|
evidence or produced earlier in the invocation.
|
||||||
|
|
||||||
|
Current provenance values:
|
||||||
|
|
||||||
Configured artifact provenance values:
|
|
||||||
- `generated.current_analyze_run`
|
- `generated.current_analyze_run`
|
||||||
- `filesystem.disabled_artifact_output`
|
- `manifest.current_analyze_artifact`
|
||||||
|
- `manifest.inputs.previous_cache`
|
||||||
|
- `current_session.previous_cache`
|
||||||
|
|
||||||
Content validation:
|
## Resolution Rules
|
||||||
- transcript built-ins: JSON with top-level `segments` array
|
|
||||||
- bounds built-in: valid JSON
|
|
||||||
- configured artifacts: non-empty text file
|
|
||||||
|
|
||||||
## Skip and resume behavior
|
Built-ins:
|
||||||
- resolver and catalog have no direct skip/resume decisions
|
|
||||||
- stage/runner skip-resume behavior consumes catalog/resolver results
|
|
||||||
|
|
||||||
## Failure behavior
|
1. manifest producer outputs (when present)
|
||||||
- unsupported source -> source validation error
|
2. canonical session-path fallback
|
||||||
- known source unavailable -> `ErrSessionArtifactNotFound`
|
|
||||||
- configured source without catalog -> resolution error
|
|
||||||
- resolved file with invalid content -> validation error
|
|
||||||
|
|
||||||
## Tests to inspect before changing
|
Configured sources (`narratio.artifact.*`):
|
||||||
- `internal/artifacts/artifact_resolver_test.go`
|
|
||||||
- `internal/artifacts/catalog_test.go`
|
|
||||||
- `internal/stage/analyze_test.go`
|
|
||||||
- `internal/config/scriptorium_test.go`
|
|
||||||
|
|
||||||
## Architectural invariants
|
- resolve only through runtime catalog availability;
|
||||||
- built-in IDs are static and registry-backed
|
- use the shared typed analyze-evidence inspection in
|
||||||
- configured artifact IDs are runtime-derived (`narratio.artifact.<name>`) and catalog-backed
|
`analyze_evidence.go` for prior current-session results;
|
||||||
- built-in/source resolution remains deterministic and validation-gated
|
- require the supported analyze-state and fingerprint versions, a `current`
|
||||||
|
record for the exact configured key and source ID, a complete contract, the
|
||||||
|
configured canonical relative path, positive stored size, and stored
|
||||||
|
checksum matching bytes read from a confined no-follow regular file; and
|
||||||
|
- treat non-current statuses, legacy or malformed records, removed keys,
|
||||||
|
unsafe or missing files, and size/checksum mismatches as unavailable without
|
||||||
|
rewriting manifest state. Catalog construction iterates current
|
||||||
|
configuration, so removed or renamed records are not advertised.
|
||||||
|
|
||||||
|
`narratio.member_artifact.*` is not a runtime source family. Configuration
|
||||||
|
resolution accepts it only in an artifact-family declaration and rewrites it
|
||||||
|
to the corresponding configured source before this catalog is built.
|
||||||
|
|
||||||
|
Prepared stable sources (`narratio.input.*`):
|
||||||
|
|
||||||
|
- resolve only from the current manifest's exact prepared-input record;
|
||||||
|
- require the policy-owned canonical path below the session root, a confined
|
||||||
|
non-symlink regular file, a non-empty payload, and a matching SHA-256
|
||||||
|
checksum; and
|
||||||
|
- return an immutable source/path/checksum/size identity shared by extract and
|
||||||
|
analyze rather than falling back to campaign/session source paths.
|
||||||
|
|
||||||
|
Extraction sources (`narratio.extraction.*`):
|
||||||
|
|
||||||
|
- use the shared typed bundle evidence inspection in `extraction_evidence.go`;
|
||||||
|
- require a current successful extract record with the exact configured source,
|
||||||
|
compatible contract and Notarius provenance, a confined regular durable
|
||||||
|
payload, matching checksum, and the current resolved trimmed-transcript
|
||||||
|
identity;
|
||||||
|
- remain unavailable unless catalog hydration receives valid evidence. Resume
|
||||||
|
treats absent or obsolete evidence as a rerun decision and unsafe evidence as
|
||||||
|
an error; and
|
||||||
|
- are never inferred by scanning the Notarius bundle directory.
|
||||||
|
|
||||||
|
Previous-session sources (`narratio.previous_session.artifact.*`):
|
||||||
|
|
||||||
|
- resolve only from local `previous/` cache state;
|
||||||
|
- prefer manifest-backed previous-input paths;
|
||||||
|
- fallback to existing previous-cache filesystem paths.
|
||||||
|
|
||||||
|
Source absence is evaluated by the consuming artifact input. An optional input
|
||||||
|
is omitted from that invocation; a required input fails resolution. This is
|
||||||
|
separate from a stage's lifecycle outcome.
|
||||||
|
|
||||||
|
Validation by content type:
|
||||||
|
|
||||||
|
- transcript JSON built-ins: JSON with top-level `segments` array;
|
||||||
|
- transcript Markdown built-ins: non-empty text file;
|
||||||
|
- bounds built-in: valid JSON;
|
||||||
|
- configured/previous-session artifact files: non-empty text file.
|
||||||
|
|
||||||
|
## Previous Requirement Collection
|
||||||
|
|
||||||
|
`CollectPreviousArtifactRequirements`:
|
||||||
|
|
||||||
|
- scans the effective configured artifact set;
|
||||||
|
- extracts only canonical previous-session sources;
|
||||||
|
- deduplicates by artifact key;
|
||||||
|
- merges required and optional references (required wins);
|
||||||
|
- returns deterministic ordering and source locations.
|
||||||
|
|
||||||
|
## Current-State Helpers
|
||||||
|
|
||||||
|
Artifacts package owns shared remote current-state loading mechanics used by
|
||||||
|
restore, status and validation checks, and previous-cache planning.
|
||||||
|
|
||||||
|
For a new-protocol current state, the pointer-selected immutable commit is the
|
||||||
|
complete restore authority. Callers receive its declared object identities and
|
||||||
|
must not supplement them by listing mutable session prefixes. The legacy reader
|
||||||
|
is intentionally separate and remains migration-only support.
|
||||||
|
|
||||||
|
The reader opens each small control object directly and enforces owner-specific
|
||||||
|
limits before decoding: 64 KiB for the mutable commit pointer, 4 MiB for the
|
||||||
|
immutable commit manifest, and 8 MiB for the selected session manifest. Legacy
|
||||||
|
compatibility applies a 4 KiB limit to `current/run_id.txt` and the same 8 MiB
|
||||||
|
manifest limit to `current/manifest.json`. These are exposed as
|
||||||
|
`MaxCurrentCommitPointerBytes`, `MaxRemoteCommitManifestBytes`,
|
||||||
|
`MaxRemoteSessionManifestBytes`, `MaxLegacyCurrentRunPointerBytes`, and
|
||||||
|
`MaxLegacyCurrentManifestBytes`.
|
||||||
|
|
||||||
|
Each read uses the generation and size metadata returned with its opened body.
|
||||||
|
Actual bytes remain subject to a limit-plus-one read even if size metadata is
|
||||||
|
absent or inaccurate. Immutable selections then retain their declared-size,
|
||||||
|
checksum, generation, and identity checks. No current-state control object is
|
||||||
|
downloaded through a temporary file.
|
||||||
|
|
||||||
|
Core helpers:
|
||||||
|
|
||||||
|
- `LoadCurrentState`
|
||||||
|
- `ValidateCurrentStateIdentity`
|
||||||
|
- `RemoteCommitManifest` and `CurrentCommitPointer`
|
||||||
|
|
||||||
|
Typed missing-state errors:
|
||||||
|
|
||||||
|
- `CurrentRunPointerMissingError` (`ErrCurrentRunPointerMissing`)
|
||||||
|
- `CurrentManifestMissingError` (`ErrCurrentManifestMissing`)
|
||||||
|
|
||||||
|
Identity validation supports caller-provided expectations:
|
||||||
|
|
||||||
|
- expected campaign;
|
||||||
|
- expected session ID;
|
||||||
|
- expected run ID, or pointer/manifest run-ID consistency check.
|
||||||
|
|
||||||
|
Caller policy is intentionally outside artifacts helpers:
|
||||||
|
|
||||||
|
- some callers fail on missing current state;
|
||||||
|
- some callers downgrade missing state to status/findings;
|
||||||
|
- some callers skip optional behavior when state is missing.
|
||||||
|
|
||||||
|
## Key Path Helpers
|
||||||
|
|
||||||
|
`internal/artifacts/paths.go` and S3-key helpers define canonical helpers for:
|
||||||
|
|
||||||
|
- session/work/run paths;
|
||||||
|
- previous-cache paths;
|
||||||
|
- spool/cache paths;
|
||||||
|
- S3 session/run/current-state key layout.
|
||||||
|
|
||||||
|
New publication creates run-scoped immutable objects, including
|
||||||
|
`runs/{run_id}/commit.json` and `runs/{run_id}/session-manifest.json`. The sole
|
||||||
|
mutable selector is `current/commit-pointer.json`; readers verify its selected
|
||||||
|
commit and declared object generations/checksums. Legacy current-pair loading
|
||||||
|
is confined to `current_state_legacy.go` for migration only.
|
||||||
|
|
||||||
|
Campaign, session, and Narratio run IDs are validated as portable opaque
|
||||||
|
segments at configuration and artifact boundaries before they can be used in a
|
||||||
|
workspace or S3 namespace. Previous-artifact destinations remain typed,
|
||||||
|
multi-segment relative paths and are confined beneath `previous/artifacts`; they
|
||||||
|
are not treated as opaque identifiers.
|
||||||
|
|
||||||
|
See [Workspace Internals](workspace.md) for how callers consume local helpers
|
||||||
|
and [Operations](../operations.md#local-state-layout) for the authoritative
|
||||||
|
physical layout.
|
||||||
|
|
||||||
|
## Invariants
|
||||||
|
|
||||||
|
- source ID formats are stable contracts;
|
||||||
|
- artifact resolution is deterministic and manifest-aware;
|
||||||
|
- extraction sources are available only from a compatible successful manifest
|
||||||
|
record;
|
||||||
|
- previous-session source resolution in `analyze` is local-only;
|
||||||
|
- remote current-state key construction remains centralized in artifacts helpers.
|
||||||
|
|
||||||
|
## Implementation And Tests
|
||||||
|
|
||||||
|
- Registry and resolution: `internal/artifacts/artifact_resolver.go`,
|
||||||
|
`internal/artifacts/catalog.go`, `internal/artifacts/transcripts.go`,
|
||||||
|
`internal/artifacts/extraction_catalog.go`,
|
||||||
|
`internal/artifacts/extraction_evidence.go`,
|
||||||
|
`internal/artifacts/extraction_input.go`,
|
||||||
|
`internal/artifacts/prepared_input.go`
|
||||||
|
- Current state: `internal/artifacts/current_state.go`,
|
||||||
|
`internal/artifacts/current_state_commit.go`,
|
||||||
|
`internal/artifacts/current_state_legacy.go`
|
||||||
|
- Paths and keys: `internal/artifacts/paths.go`,
|
||||||
|
`internal/artifacts/s3_keys.go`
|
||||||
|
- Previous requirements: `internal/artifacts/previous_requirements.go`
|
||||||
|
- Tests: `internal/artifacts/artifact_resolver_test.go`,
|
||||||
|
`internal/artifacts/catalog_test.go`,
|
||||||
|
`internal/artifacts/extraction_catalog_test.go`,
|
||||||
|
`internal/artifacts/current_state_test.go`,
|
||||||
|
`internal/artifacts/paths_model_test.go`,
|
||||||
|
`internal/artifacts/previous_requirements_test.go`
|
||||||
|
|||||||
110
docs/internal/command-restore.md
Normal file
110
docs/internal/command-restore.md
Normal file
@@ -0,0 +1,110 @@
|
|||||||
|
# Internal: Command Restore
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
|
||||||
|
Explain the implemented restore discovery, planning, installation, and
|
||||||
|
reporting flow in `internal/app`. User invocation belongs in
|
||||||
|
[CLI](../cli.md#session-restore), and the operator recovery procedure and
|
||||||
|
physical restore scope belong in
|
||||||
|
[Operations](../operations.md#restore-workflow).
|
||||||
|
|
||||||
|
Restore separates remote authority, local conflict policy, and filesystem
|
||||||
|
mutation so each remains testable independently.
|
||||||
|
|
||||||
|
## Discovery Contract
|
||||||
|
|
||||||
|
Discovery delegates current-state pointer and manifest loading to
|
||||||
|
`internal/artifacts`, then validates the result against the resolved request:
|
||||||
|
|
||||||
|
- campaign must match;
|
||||||
|
- session ID must match.
|
||||||
|
- run ID must match the pointer-selected committed run.
|
||||||
|
|
||||||
|
Restore treats any missing or invalid remote current state as a command error.
|
||||||
|
|
||||||
|
## Planning Contract
|
||||||
|
|
||||||
|
Restore planner action kinds:
|
||||||
|
|
||||||
|
- `download`;
|
||||||
|
- `skip_same`;
|
||||||
|
- `conflict`.
|
||||||
|
|
||||||
|
Planner behavior:
|
||||||
|
|
||||||
|
- a new-protocol restore uses only the selected commit's declared artifact set;
|
||||||
|
each action carries that artifact's immutable key, checksum, size, and
|
||||||
|
generation. Coherent legacy state remains on the isolated compatibility path;
|
||||||
|
- remote-to-local mapping is traversal-safe;
|
||||||
|
- actions are sorted by local relative path and then remote key;
|
||||||
|
- force converts differing eligible regular files from conflicts to downloads;
|
||||||
|
directories and other non-regular targets remain conflicts.
|
||||||
|
|
||||||
|
For a non-dry-run restore, planning/classification happens only after acquiring
|
||||||
|
the session lock. Runner manifest/reuse checks acquire that same lock first.
|
||||||
|
|
||||||
|
Previous-cache readiness is resolved through `previouscache.Resolve` for restore,
|
||||||
|
prepare, status, and validation. A committed source is selected only by its
|
||||||
|
exact source identity; legacy fallback remains isolated and rejects ambiguity.
|
||||||
|
|
||||||
|
## Execution Contract
|
||||||
|
|
||||||
|
Execution order and safety:
|
||||||
|
|
||||||
|
- non-manifest downloads happen before manifest install;
|
||||||
|
- `manifest.json` installs last;
|
||||||
|
- downloads use sibling temp files plus atomic rename;
|
||||||
|
- manifest replacement is validated before rename;
|
||||||
|
- each committed object is verified against its declared checksum, size, and
|
||||||
|
generation before installation;
|
||||||
|
- a committed manifest already verified during discovery is retained for the
|
||||||
|
matching restore action and revalidated before installation, avoiding a
|
||||||
|
second body transfer;
|
||||||
|
- failed installs do not roll back files already written in the same execution.
|
||||||
|
- a durable `.restore-incomplete.json` marker is written before installation.
|
||||||
|
It blocks runners until a restore retry completes all verified installs and
|
||||||
|
the local manifest replacement, at which point it is removed.
|
||||||
|
- restored manifest local references are rebased beneath the selected local
|
||||||
|
session root. Unsafe relative references and producer-machine absolute paths
|
||||||
|
outside the manifest's producer session root are rejected; producer-local
|
||||||
|
spool/cache and cleanup locations are not restored as authority.
|
||||||
|
|
||||||
|
Audio restore path:
|
||||||
|
|
||||||
|
- uses `audio.MaterializeS3Audio`;
|
||||||
|
- integrates spool and S3 audio cache paths;
|
||||||
|
- reuses cached audio only when its no-follow regular file, content digest, and
|
||||||
|
identity sidecar all match the selected remote object version; otherwise it
|
||||||
|
refreshes through the durable download path.
|
||||||
|
|
||||||
|
## Reporting Contract
|
||||||
|
|
||||||
|
- dry-run mode prints a summary, performs no durable session writes, and may
|
||||||
|
read remote current-state or object-identity data to produce that summary;
|
||||||
|
- execution mode persists the canonical restore report described in
|
||||||
|
[Operations](../operations.md#restore-workflow);
|
||||||
|
- report includes plan counts, per-action status, and execution failures.
|
||||||
|
|
||||||
|
## Invariants
|
||||||
|
|
||||||
|
- restore uses committed remote current state as authority;
|
||||||
|
- one restore or status inspection observes the single pointer-selected commit
|
||||||
|
loaded at discovery; later pointer changes cannot add objects or substitute a
|
||||||
|
different run into its plan;
|
||||||
|
- a verified `current/commit-pointer.json` and its selected immutable commit
|
||||||
|
establish new-protocol remote commitment; coherent legacy
|
||||||
|
`current/run_id.txt` plus `current/manifest.json` remains read-only migration
|
||||||
|
support;
|
||||||
|
- restore does not execute pipeline stages.
|
||||||
|
|
||||||
|
## Implementation And Tests
|
||||||
|
|
||||||
|
- Discovery: `internal/app/restore_discovery.go`
|
||||||
|
- Planning: `internal/app/restore_plan.go`, `internal/previouscache`
|
||||||
|
- Execution: `internal/app/restore_execute.go`
|
||||||
|
- Reporting and command coordination: `internal/app/restore_report.go`,
|
||||||
|
`internal/app/restore.go`
|
||||||
|
- Tests: `internal/app/restore_discovery_test.go`,
|
||||||
|
`internal/app/restore_plan_test.go`,
|
||||||
|
`internal/app/restore_execution_test.go`,
|
||||||
|
`internal/app/restore_workflow_test.go`
|
||||||
180
docs/internal/configuration.md
Normal file
180
docs/internal/configuration.md
Normal file
@@ -0,0 +1,180 @@
|
|||||||
|
# Configuration Internals
|
||||||
|
|
||||||
|
User-visible fields, defaults, and selection behavior belong in the
|
||||||
|
[Configuration Reference](../config.md). This document describes the internal
|
||||||
|
pipeline-loading boundary implemented by `internal/config`.
|
||||||
|
|
||||||
|
## Pipeline Loading
|
||||||
|
|
||||||
|
`LoadPipeline` assembles and validates a pipeline in this order:
|
||||||
|
|
||||||
|
1. Parse the root YAML into a presence-aware composition tree. The tree retains
|
||||||
|
source names, full field paths, node kinds, declaration order, and explicit
|
||||||
|
zero, false, empty-map, and empty-list values.
|
||||||
|
2. Remove the root-only `composition` envelope and validate its explicit
|
||||||
|
`imports`, `default_profile`, and named `profiles` declarations. A load
|
||||||
|
option retains the difference between omitted and explicitly empty profile
|
||||||
|
selection.
|
||||||
|
3. Open each import relative to the root pipeline directory through the
|
||||||
|
confined regular-file boundary. Imports must use a `.yml` or `.yaml`
|
||||||
|
extension and cannot traverse, use symlinks, repeat a file, import the root,
|
||||||
|
or contain another composition envelope.
|
||||||
|
4. Resolve and structurally parse every declared profile overlay through the
|
||||||
|
same confined regular-file boundary. Missing or malformed unselected
|
||||||
|
overlays fail the load. Overlays cannot contain a composition envelope.
|
||||||
|
5. Additively merge the root body and imports. Distinct map leaves compose;
|
||||||
|
repeated scalar or list paths and node-kind disagreements are conflicts.
|
||||||
|
6. Select exactly one declared profile from an explicit option or the default,
|
||||||
|
then recursively merge its overlay. Overlay leaves replace base leaves,
|
||||||
|
lists are atomic replacements, and null or kind changes fail.
|
||||||
|
7. Emit deterministic canonical YAML and strictly decode it into
|
||||||
|
`PipelineConfig`.
|
||||||
|
8. Apply pipeline defaults once, resolve ordinary relative pipeline paths from
|
||||||
|
the root pipeline file, and digest the normalized effective mapping.
|
||||||
|
|
||||||
|
This ordering preserves monolithic configuration behavior. Moving a field to
|
||||||
|
an imported fragment changes its source ownership, not its path base, default,
|
||||||
|
or schema semantics.
|
||||||
|
|
||||||
|
## Loaded Context Resolution
|
||||||
|
|
||||||
|
`LoadedPipelineCampaign` carries one already composed pipeline and its selected
|
||||||
|
campaign into session resolution. `LoadSessionWithPipelineCampaignOptions`
|
||||||
|
loads a local session against that context, while
|
||||||
|
`ResolveLoadedPipelineCampaign` also accepts an already loaded remote session
|
||||||
|
or no session while a caller retrieves one. Compatibility loaders route through
|
||||||
|
these functions after their initial pipeline and campaign reads.
|
||||||
|
|
||||||
|
Application commands own pipeline and campaign discovery, campaign-file versus
|
||||||
|
registry selection, and the corresponding mutual-exclusion rules. Once they
|
||||||
|
have a `LoadedPipelineCampaign`, local session discovery and remote-session
|
||||||
|
download retain that exact pipeline object and its private provenance. Removing
|
||||||
|
a temporary downloaded session file therefore cannot invalidate the resolved
|
||||||
|
pipeline or campaign context.
|
||||||
|
|
||||||
|
The application also has a separate read-only inspection resolver for `config
|
||||||
|
validate`, `config show`, and `config sources`. It uses the same production root/profile and
|
||||||
|
campaign selection functions, but never routes through session discovery,
|
||||||
|
remote-session download, secret loading, adapter composition, workspace
|
||||||
|
initialization, manifest access, or cleanup. A pipeline with retained artifact
|
||||||
|
family declarations must resolve its selected campaign before ordinary pipeline
|
||||||
|
validation, which expands its canonical-party members and generated publish
|
||||||
|
rules. A pipeline without those declarations may be validated by itself.
|
||||||
|
|
||||||
|
`MarshalEffectivePipeline` is the configuration-owned projection for `config
|
||||||
|
show`. It serializes the typed, defaulted effective mapping through the
|
||||||
|
deterministic composition renderer, then removes resolution-only artifact
|
||||||
|
family declarations. The result contains no composition envelope or private
|
||||||
|
provenance fields and has one trailing newline; commands do not marshal runtime
|
||||||
|
objects directly.
|
||||||
|
|
||||||
|
`EffectivePipelineSources` and `EffectiveCampaignSources` provide the separate
|
||||||
|
safe provenance projection for `config sources`. Pipeline ownership begins with
|
||||||
|
the complete logical field paths retained during composition and classifies
|
||||||
|
each contributor as root, import, profile, or centralized default. The
|
||||||
|
projection replaces generated concrete member paths with paired family and
|
||||||
|
canonical-party records, and does the same for generated publish rules.
|
||||||
|
Campaign records identify campaign-owned fields and party inputs; canonical
|
||||||
|
derived players point to the party source, while legacy players retain a
|
||||||
|
dedicated legacy-player role. The application command only joins these sorted
|
||||||
|
records with selection metadata and never reparses configuration files.
|
||||||
|
|
||||||
|
`config diff` uses a paired profile loader that parses the root, imports, and
|
||||||
|
declared overlays once, then clones the additive base before independently
|
||||||
|
selecting, decoding, defaulting, and finalizing each profile. When campaign
|
||||||
|
resolution is needed, the command loads one selected campaign and party and
|
||||||
|
expands both effective pipelines from that same party value. The configuration
|
||||||
|
owner projects each normalized effective mapping into sorted logical paths;
|
||||||
|
mapping leaves are compared individually while sequence values remain atomic.
|
||||||
|
Values are compact deterministic JSON representations for command output, not
|
||||||
|
raw YAML fragments, ownership records, or secret material. A differing digest
|
||||||
|
with no projected difference is treated as an internal consistency error.
|
||||||
|
|
||||||
|
Campaign context construction also reads and classifies the campaign-owned
|
||||||
|
party source through `ParseParty`. A canonical party retains its raw bytes and
|
||||||
|
normalized roster in runtime-only `ResolvedParty` provenance, while a legacy
|
||||||
|
party remains opaque. Canonical resolution creates a virtual
|
||||||
|
`derived_from_party` players input and rejects competing campaign or session
|
||||||
|
players files and session party overrides. The compact legacy compatibility
|
||||||
|
path resolves the effective campaign/session party and players files together.
|
||||||
|
|
||||||
|
## Canonical Party Domain
|
||||||
|
|
||||||
|
`ParseParty` is the package-owned boundary for classifying a party source.
|
||||||
|
When a top-level `schema_version` is present, it strictly validates the
|
||||||
|
`narratio.party.v1` contract into ordered character domain values. The
|
||||||
|
canonical value retains a separate exact byte copy of its source so consumers
|
||||||
|
can materialize the authored party document without reserializing it. Its
|
||||||
|
`PlayersYAML` method deterministically derives the versioned players-only
|
||||||
|
projection.
|
||||||
|
|
||||||
|
An unversioned source is classified by the small legacy compatibility boundary
|
||||||
|
in `party_legacy.go`; it deliberately exposes no parsed roster information.
|
||||||
|
That boundary exists solely to isolate removable compatibility behavior from
|
||||||
|
the canonical parser.
|
||||||
|
|
||||||
|
## Diagnostics And Runtime Metadata
|
||||||
|
|
||||||
|
Syntax, duplicate-key, composition, conflict, and schema failures include the
|
||||||
|
relevant source name and full field path. Additive conflicts report every
|
||||||
|
claiming source so operators can repair the split without repeatedly
|
||||||
|
rediscovering additional conflicts.
|
||||||
|
|
||||||
|
The loaded pipeline retains private runtime metadata for the absolute root
|
||||||
|
path, ordered imports, selected profile name and selection source, selected
|
||||||
|
overlay, contributing sources, effective digest, and leaf ownership. Base
|
||||||
|
leaves retain their root/import owners, replaced leaves belong to the selected
|
||||||
|
overlay, and centrally supplied values use the synthetic `default` owner. This
|
||||||
|
metadata does not participate in YAML decoding or alter the public
|
||||||
|
configuration model.
|
||||||
|
|
||||||
|
The effective digest is SHA-256 over deterministic canonical YAML produced from
|
||||||
|
the defaulted `PipelineConfig`. Runtime Notarius paths remain absolute for
|
||||||
|
execution, but the digest substitutes their normalized logical values captured
|
||||||
|
before root-relative resolution, so relocating an equivalent configuration
|
||||||
|
bundle does not change provenance. Because composition and resolution metadata
|
||||||
|
are private, the digest excludes source layout, profile name, and ownership.
|
||||||
|
Configuration stores environment variable names rather than resolving raw
|
||||||
|
credentials, so raw secret values are neither loaded nor hashed.
|
||||||
|
`recomputePipelineEffectiveDigest` is the single package-owned refresh point
|
||||||
|
for later runtime expansion.
|
||||||
|
|
||||||
|
## Test Surfaces
|
||||||
|
|
||||||
|
`composition_test.go` protects the presence and merge algebra independently of
|
||||||
|
the public schema. `pipeline_composition_test.go` exercises explicit imports,
|
||||||
|
confinement, conflicts, strict decoding, metadata, and root-relative path
|
||||||
|
behavior through `LoadPipeline`. `pipeline_profiles_test.go` covers selection,
|
||||||
|
all-overlay validation, overlay behavior, provenance, option propagation, and
|
||||||
|
effective-digest stability. Application configuration-loader tests protect the
|
||||||
|
single-read boundary by changing the pipeline file after its initial load and
|
||||||
|
confirming local session resolution retains the original pipeline. Other
|
||||||
|
configuration tests continue to protect defaults and validation after assembly.
|
||||||
|
`party_test.go` protects the versioned party schema, domain invariants, and
|
||||||
|
deterministic players projection without involving campaign or runtime wiring.
|
||||||
|
`party_resolution_test.go` protects campaign-owned party loading, canonical
|
||||||
|
input restrictions, legacy overrides, source provenance, and virtual players
|
||||||
|
input selection.
|
||||||
|
|
||||||
|
## Artifact Family Resolution
|
||||||
|
|
||||||
|
Pipeline loading retains `scriptorium.artifact_families` as a resolution-only
|
||||||
|
declaration. Once campaign party resolution establishes a canonical roster,
|
||||||
|
configuration expands families in sorted family-key and character-ID order
|
||||||
|
into ordinary `ScriptoriumArtifactConfig` values. The expansion owns the narrow
|
||||||
|
`{character_id}` output substitution, closed member-variable selectors, key and
|
||||||
|
output collision checks, and the runtime-only family-origin catalog. It then
|
||||||
|
removes family declarations from `ScriptoriumConfig`, runs ordinary Scriptorium
|
||||||
|
validation, and refreshes the effective pipeline digest. Stages and adapters
|
||||||
|
therefore receive only concrete artifact maps.
|
||||||
|
|
||||||
|
The catalog retains sorted family member keys plus family/character/source
|
||||||
|
origins and the typed dependency/publish declarations for their later owners.
|
||||||
|
`member_dependencies` add corresponding ordinary concrete dependencies, while
|
||||||
|
the family-only `narratio.member_artifact.<family>` input form is rewritten to
|
||||||
|
the matching ordinary configured-artifact source. The catalog records those
|
||||||
|
resolved dependency and input identities with their declaring family and party
|
||||||
|
member. No member-artifact source is registered as a runtime policy source.
|
||||||
|
An enabled family publish declaration expands to ordinary configured-artifact
|
||||||
|
publish rules before the existing publish and lock validators run. Runtime
|
||||||
|
publication consequently receives no family wildcard or special matcher.
|
||||||
71
docs/internal/fileops.md
Normal file
71
docs/internal/fileops.md
Normal file
@@ -0,0 +1,71 @@
|
|||||||
|
# Internal: File Operations
|
||||||
|
|
||||||
|
`internal/fileops` owns the narrow mechanics for durable replacement of one
|
||||||
|
byte file. Callers keep ownership of serialization, validation, cancellation,
|
||||||
|
and destination-directory policy.
|
||||||
|
|
||||||
|
## Destination Confinement
|
||||||
|
|
||||||
|
Before it creates, replaces, or installs a destination file, `fileops` opens
|
||||||
|
each ancestor from the filesystem root and rejects symbolic links or components
|
||||||
|
that change during traversal. The resulting parent-directory handle is retained
|
||||||
|
for sibling temporary-file creation and rename, so a later pathname swap cannot
|
||||||
|
redirect the replacement. Existing destination symlinks are replaced as leaf
|
||||||
|
entries; their targets are never followed.
|
||||||
|
|
||||||
|
Remote object acquisition uses a writer supplied by the storage owner. The
|
||||||
|
writer receives a `fileops`-owned, already-open sibling temporary file rather
|
||||||
|
than a mutable destination path. Callers still own remote object selection,
|
||||||
|
validation, conflict handling, and final mode.
|
||||||
|
|
||||||
|
Directory promotion keeps the verified destination parent open while it creates
|
||||||
|
the temporary tree, copies regular source entries, and performs the platform
|
||||||
|
no-replace rename. Platforms without a verified handle-relative atomic
|
||||||
|
no-replace primitive reject promotion before writing a temporary tree.
|
||||||
|
|
||||||
|
## Cleanup Contract
|
||||||
|
|
||||||
|
`RemoveAllUnderRoot` accepts an explicit root and a proper descendant. It opens
|
||||||
|
the root and each target ancestor without following symlinks, then removes the
|
||||||
|
tree through those directory handles. It rejects root deletion and any symlink
|
||||||
|
encountered in the target path or tree; repeated removal of a missing target is
|
||||||
|
successful. Command and post-publish policy remains owned by `internal/app`.
|
||||||
|
|
||||||
|
## Confined Reads
|
||||||
|
|
||||||
|
`ReadRegularFileUnderRoot` is the no-follow, bounded read primitive for a
|
||||||
|
caller-selected root and relative file path; `ReadRegularFile` is its
|
||||||
|
path-based convenience wrapper. They verify every ancestor through directory
|
||||||
|
handles and admit only a stable regular-file handle. Callers enforce their own
|
||||||
|
byte limits and access policy. Credential mode policy and environment
|
||||||
|
precedence remain owned by `internal/app`.
|
||||||
|
|
||||||
|
## Replacement Contract
|
||||||
|
|
||||||
|
`ReplaceFileAtomic` requires an existing destination directory. It creates a
|
||||||
|
sibling temporary file, writes the complete byte sequence, applies the
|
||||||
|
caller-supplied mode, syncs and closes the file, runs an optional pre-rename
|
||||||
|
check, replaces the destination with a rename, then syncs the containing
|
||||||
|
directory.
|
||||||
|
|
||||||
|
The pre-rename check is the last point at which a caller can cancel without
|
||||||
|
installing a new destination. A failure before the rename leaves the old
|
||||||
|
destination unchanged and removes the temporary file; any cleanup failure is
|
||||||
|
returned alongside the primary failure. A failure after the rename may leave
|
||||||
|
the new file visible, but it is not reported as crash-durable.
|
||||||
|
|
||||||
|
Replacement follows the operating system's same-filesystem rename semantics.
|
||||||
|
If a platform cannot replace an existing destination, the operation returns an
|
||||||
|
error and never removes the old file as an emulation step.
|
||||||
|
|
||||||
|
## Directory-Sync Support
|
||||||
|
|
||||||
|
Linux and macOS attempt to sync the destination directory. Windows opens the
|
||||||
|
directory with backup semantics and flushes its buffers. If either operation
|
||||||
|
is unavailable for the platform, directory handle, or filesystem,
|
||||||
|
`ErrDirectorySyncUnsupported` is returned. Narratio does not treat that result
|
||||||
|
as successful crash-durable replacement.
|
||||||
|
|
||||||
|
`WriteFileAtomic`, copy helpers, and downloaded temporary-file installation
|
||||||
|
retain their compatibility behavior of creating the destination parent with
|
||||||
|
the repository's workspace permissions before using this contract.
|
||||||
@@ -1,81 +1,316 @@
|
|||||||
# Internal: Manifest
|
# Internal: Manifest
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Describe Narratio's durable execution state model for session-level and run-level manifests, including lifecycle transitions and persistence behavior.
|
|
||||||
|
|
||||||
## Inputs and outputs
|
Explain the session-progress and invocation-audit models implemented by
|
||||||
Inputs:
|
`internal/manifest`. Physical manifest placement belongs in
|
||||||
- Session identity and run identity from app orchestration.
|
[Operations](../operations.md#local-state-layout).
|
||||||
- Stage transition events and stage result payloads.
|
|
||||||
|
|
||||||
Outputs:
|
## Session Manifest
|
||||||
- Session manifest at `{workspace.root}/work/{campaign}/{session_id}/manifest.json`.
|
|
||||||
- Run manifest at `{workspace.root}/work/{campaign}/{session_id}/runs/{run_id}/manifest.json`.
|
|
||||||
|
|
||||||
## Boundaries
|
`manifest.Manifest` records:
|
||||||
Owns:
|
|
||||||
- Manifest schemas (`Manifest`, `RunManifest`, stage records, error records, input/artifact records).
|
|
||||||
- Stage status/action transition methods.
|
|
||||||
- Persistent store contract (`manifest.Store`) and local JSON store implementation.
|
|
||||||
|
|
||||||
Does not own:
|
- identity (`session_id`, `campaign`, `run_id`)
|
||||||
- Stage implementation details.
|
- local path metadata (`local_workdir`, `local_spool_dir`)
|
||||||
- Path construction policy outside manifest file persistence calls.
|
- remote identity metadata (`s3_bucket`, `s3_session_prefix`, `s3_run_prefix`)
|
||||||
- CLI command behavior.
|
- `inputs` records
|
||||||
|
- durable `artifacts` records
|
||||||
|
- per-stage `stages` map
|
||||||
|
- an optional `post_publish_cleanup` obligation, which binds a committed run,
|
||||||
|
remote commit identity, and each exact root-confined local target to its
|
||||||
|
completion evidence
|
||||||
|
|
||||||
## Config fields used
|
Session, campaign, and run identities in local and downloaded manifests must be
|
||||||
Manifest package itself does not read config directly.
|
portable opaque segments. Unsafe legacy identities are rejected with migration
|
||||||
|
guidance rather than being normalized into a different workspace or remote
|
||||||
|
namespace.
|
||||||
|
|
||||||
Manifest identity fields are populated by app/stage orchestration from:
|
Prepare records independent `party` and `players` input checksums. In canonical
|
||||||
- `session.session_id`
|
party mode, the party record retains its campaign source identity while the
|
||||||
- `session.campaign`
|
players record uses `derived_from_party`; raw roster content is never embedded
|
||||||
- `pipeline.workspace.root`
|
in manifest metadata. Both records remain the durable authority for consumers
|
||||||
- `pipeline.storage.s3.*` (when archive/S3 identity is set)
|
of their prepared input source IDs.
|
||||||
|
|
||||||
## External adapters used
|
The model admits these stage states:
|
||||||
- No external service adapters.
|
|
||||||
- Uses local filesystem for persistence via `manifest.LocalStore`.
|
|
||||||
|
|
||||||
## State and manifest behavior
|
- `pending`
|
||||||
Session manifest model:
|
- `running`
|
||||||
- Tracks durable per-session stage state and provenance (`pending`, `running`, `succeeded`, `failed`, `skipped`, `stale`, `interrupted`).
|
- `succeeded`
|
||||||
- Stores resolved inputs, durable artifacts, stage logs/config refs, and stage metadata.
|
- `failed`
|
||||||
|
- `skipped`
|
||||||
|
- `stale`
|
||||||
|
- `interrupted`
|
||||||
|
|
||||||
Run manifest model:
|
### Analyze-owned artifact state
|
||||||
- Tracks one invocation (`run_id`) with requested stages and force mode.
|
|
||||||
- Tracks per-stage action (`run` or `skip`) and per-stage status.
|
|
||||||
- Tracks overall run status (`running`, `succeeded`, `failed`).
|
|
||||||
|
|
||||||
Persistence behavior:
|
The `analyze` stage record may carry `analyze_state_version: 1` and an
|
||||||
- Load validates required identity/timestamp fields and normalizes maps/records.
|
`analyze_artifacts` map keyed by normalized configured artifact key. The
|
||||||
- Save updates `updated_at` and writes JSON atomically (temp file + rename).
|
version is the authority marker: version 1 with no entries is a valid evaluated
|
||||||
- Session and run manifests are saved incrementally before/after stage transitions.
|
empty set, while an absent version is legacy aggregate-only state and provides
|
||||||
|
no current configured-artifact evidence.
|
||||||
|
|
||||||
Relationship during execution:
|
Each analyze artifact record has one disposition:
|
||||||
- Runner updates both manifests for every stage transition.
|
|
||||||
- Session manifest is the durable pipeline-progress ledger.
|
|
||||||
- Run manifest is invocation history and audit record.
|
|
||||||
- Analyze stage outputs are persisted as `kind=scriptorium_artifact` with `source_id=narratio.artifact.<name>` for configured artifact identity.
|
|
||||||
|
|
||||||
## Skip and resume behavior
|
- `current`: the configured artifact is available and carries a versioned
|
||||||
- Resume and skip decisions are based on session-manifest stage statuses.
|
fingerprint plus a complete output record and separate output size;
|
||||||
- `--force` reruns selected stages and marks downstream succeeded stages as `stale` in session manifest.
|
- `stale`: the recorded semantic identity is no longer current;
|
||||||
- Run manifest records whether each stage was executed or skipped in that invocation.
|
- `missing`: no validated current result exists;
|
||||||
|
- `failed`: the attempted work failed and carries a bounded diagnostic; or
|
||||||
|
- `unselected`: the artifact was intentionally outside the evaluated set.
|
||||||
|
|
||||||
## Failure behavior
|
Records bind their normalized key and dependencies, fingerprint contract when
|
||||||
- Stage failure marks both manifests failed for that stage and records error messages/timestamps.
|
evaluated, canonical session-relative output identity when current, producing
|
||||||
- Save failures are returned immediately and fail the command.
|
Narratio run, update time, and bounded non-secret Scriptorium provenance and
|
||||||
- Invalid/malformed manifest files fail load with explicit validation/decode errors.
|
diagnostic paths. A current output includes its configured source ID, contract,
|
||||||
|
checksum, and positive byte size. Non-current records cannot carry an output,
|
||||||
|
so an older file is not advertised through stale, missing, failed, or
|
||||||
|
unselected state.
|
||||||
|
|
||||||
## Tests to inspect before changing
|
Family-produced records additionally retain optional `family` and
|
||||||
- `internal/manifest/manifest_test.go`
|
`character_id` provenance supplied by configuration resolution. These fields
|
||||||
- `internal/manifest/run_manifest_test.go`
|
do not replace the concrete configured key or infer family membership from a
|
||||||
- `internal/manifest/store_test.go`
|
name, so older records without them remain valid.
|
||||||
- `internal/app/runner_test.go`
|
|
||||||
- `internal/app/run_control_test.go`
|
|
||||||
- `internal/app/resume_run_stage_test.go`
|
|
||||||
|
|
||||||
## Architectural invariants
|
The session-stage collection is the reconciled authority across invocations.
|
||||||
- Session manifest is authoritative for stage progression across invocations.
|
The corresponding collection on an invocation's `analyze` stage record is an
|
||||||
- Run manifest is invocation-scoped and never replaces session manifest as progress authority.
|
audit of only the artifacts evaluated or attempted by that run. These records
|
||||||
- Manifest writes are atomic and deterministic (JSON + newline, temp rename pattern).
|
remain analyze-owned data inside the fixed stage; they are not dynamic stages
|
||||||
|
or generic subtasks.
|
||||||
|
|
||||||
|
The stage result contract has one analyze-specific projection boundary. On
|
||||||
|
success, the runner validates and deep-copies the complete reconciled session
|
||||||
|
collection and the invocation subset. Aggregate session outputs are rebuilt in
|
||||||
|
configured-key order from current session records only; invocation outputs are
|
||||||
|
limited to current records produced by that invocation's run ID. Ordinary
|
||||||
|
stage outputs cannot accompany this projection, so there is one source of
|
||||||
|
artifact authority.
|
||||||
|
|
||||||
|
Successful incremental execution replaces only evaluated artifact records and
|
||||||
|
preserves valid unrelated current records. Rebuilt outputs are compared by
|
||||||
|
bytes and contract: an unchanged identity permits an unselected dependent with
|
||||||
|
the same recomputed fingerprint to remain current, while a changed identity
|
||||||
|
removes output authority from every unselected transitive dependent by marking
|
||||||
|
it stale. A partial analyze invocation can therefore succeed while unrelated
|
||||||
|
configured records remain stale. Existing canonical files never create current
|
||||||
|
records without validated execution and projection.
|
||||||
|
|
||||||
|
Aggregate analyze status is deliberately coarser than this collection. Resume
|
||||||
|
validation may skip a succeeded aggregate record when the selected artifact
|
||||||
|
closure is current even if unrelated records are stale. Conversely, a stale
|
||||||
|
aggregate record may cross the ordinary runner boundary and perform zero
|
||||||
|
Scriptorium calls when reconciliation proves every selected artifact current;
|
||||||
|
the successful projection then restores the aggregate status.
|
||||||
|
|
||||||
|
Analyze may return a projection together with an error. That restricted result
|
||||||
|
cannot carry ordinary outputs, skip state, aggregate logs, generated configs,
|
||||||
|
or metadata. The runner persists only the validated per-artifact collections,
|
||||||
|
then marks the aggregate analyze and run state failed and invalidates delivery
|
||||||
|
dependents conservatively. Unrelated current records survive because the
|
||||||
|
session projection is complete. A malformed projection is not applied, and a
|
||||||
|
failed session projection save restores the prior per-artifact authority before
|
||||||
|
terminal failure persistence.
|
||||||
|
|
||||||
|
The incremental executor constructs this restricted projection at each
|
||||||
|
scheduled artifact boundary. The active record is failed without output,
|
||||||
|
current transitive dependents are stale, unrelated current records survive, and
|
||||||
|
only earlier validated and materialized completions remain current in the
|
||||||
|
invocation subset. Session failure state is persisted before invocation failure
|
||||||
|
state. If either terminal save fails, its persistence error is joined with the
|
||||||
|
original adapter, validation, or filesystem cause; a failed projection save
|
||||||
|
does not turn incidental canonical bytes into manifest authority.
|
||||||
|
|
||||||
|
## Run Manifest
|
||||||
|
|
||||||
|
`manifest.RunManifest` is created for each invocation and records:
|
||||||
|
|
||||||
|
- invocation identity and `force` flag
|
||||||
|
- the selected profile (when any) and secret-free effective configuration digest
|
||||||
|
- requested stages
|
||||||
|
- per-stage action (`run` or `skip`)
|
||||||
|
- per-stage status
|
||||||
|
- overall run status (`running`, `succeeded`, `failed`)
|
||||||
|
|
||||||
|
## Remote Commit Manifest
|
||||||
|
|
||||||
|
`artifacts.RemoteCommitManifest` is a separate, versioned remote snapshot
|
||||||
|
contract. It is not a serialized session manifest and contains no local
|
||||||
|
post-publication assertion such as `current_pointer_written`. A remote commit
|
||||||
|
identifies one campaign, session, and run and declares its immutable artifact
|
||||||
|
set. Each artifact has a typed source, immutable destination key, SHA-256
|
||||||
|
checksum, size, and storage generation.
|
||||||
|
|
||||||
|
`current/commit-pointer.json` is the sole mutable selector for the new
|
||||||
|
contract. It identifies exactly one run-scoped `runs/{run_id}/commit.json` and
|
||||||
|
binds that object by checksum, size, and generation. Readers strictly reject
|
||||||
|
unknown fields, version mismatches, pointer/commit identity mismatches, and
|
||||||
|
objects that do not match their declaration.
|
||||||
|
|
||||||
|
The reader retains a temporary, clearly isolated compatibility path for a
|
||||||
|
coherent legacy `current/manifest.json` plus `current/run_id.txt` pair. That
|
||||||
|
path is removable after migration and is never used to write new state.
|
||||||
|
|
||||||
|
## Persistence Semantics
|
||||||
|
|
||||||
|
`manifest.LocalStore`:
|
||||||
|
|
||||||
|
- validates loaded documents;
|
||||||
|
- normalizes missing maps/stage records;
|
||||||
|
- writes through a sibling temporary file, syncing the completed file and
|
||||||
|
destination directory after atomic replacement;
|
||||||
|
- updates `updated_at` on save.
|
||||||
|
|
||||||
|
If the operating system or filesystem cannot sync a directory, save returns an
|
||||||
|
explicit error instead of claiming crash-durable replacement. A returned error
|
||||||
|
after the rename can therefore leave the new manifest visible but not confirmed
|
||||||
|
durable; callers must reload it before retrying.
|
||||||
|
|
||||||
|
## Execution Semantics
|
||||||
|
|
||||||
|
The application runner marks an executing stage running and then succeeded or
|
||||||
|
failed in both manifests, persisting each transition. On success it records
|
||||||
|
outputs, logs, generated configuration references, metadata, and—when the
|
||||||
|
stage implements the optional contract—a versioned semantic-configuration
|
||||||
|
fingerprint. Artifact
|
||||||
|
records may include optional contract and external provenance objects; old
|
||||||
|
manifests remain compatible when those fields are absent. A successful forced
|
||||||
|
rerun marks only succeeded transitive dependent session-stage records stale.
|
||||||
|
The application owns a fixed dependency relation distinct from execution order;
|
||||||
|
dependents are returned in canonical order. Render and extract therefore never
|
||||||
|
stale one another, while either can stale analyze, publish, and notify.
|
||||||
|
|
||||||
|
Starting an execution clears the current session-stage record's prior outputs,
|
||||||
|
logs, generated configuration references, metadata, and semantic fingerprint.
|
||||||
|
Failed and skipped
|
||||||
|
transitions enforce the same clearing rule directly, while success repopulates
|
||||||
|
only fields returned by the new result. Marking a record stale does not clear
|
||||||
|
those details because resume validation and diagnosis may still require them
|
||||||
|
before execution begins. Invocation run manifests remain immutable audit
|
||||||
|
records of their own outcomes.
|
||||||
|
|
||||||
|
Aggregate lifecycle clearing deliberately preserves the analyze-owned
|
||||||
|
per-artifact collection. This lets later reconciliation replace only evaluated
|
||||||
|
entries without erasing unrelated current results. Other stages retain their
|
||||||
|
existing aggregate-only lifecycle behavior and are forbidden from carrying the
|
||||||
|
analyze-specific fields.
|
||||||
|
|
||||||
|
A stage may explicitly return a skipped disposition and stable reason. The
|
||||||
|
runner persists that outcome in both manifests, clears older outputs for the
|
||||||
|
session-stage record along with older logs, generated configuration references,
|
||||||
|
and metadata, then applies any bounded details from the current skip and
|
||||||
|
continues. This self-skip is distinct from deciding not to execute an
|
||||||
|
already-succeeded stage and is reconsidered on later runs. Skipped results
|
||||||
|
cannot contain outputs. An intentional self-skip records the current semantic
|
||||||
|
fingerprint because it is a completed, reusable stage result; failed or
|
||||||
|
interrupted work never promotes one.
|
||||||
|
|
||||||
|
When an already-succeeded stage is skipped, the invocation run manifest records
|
||||||
|
the `skip` action and reason. The session manifest deliberately retains its
|
||||||
|
existing succeeded record because it remains the cross-invocation progress
|
||||||
|
authority. If a stage supplies semantic configuration evidence, reuse first
|
||||||
|
requires the persisted positive schema version and lowercase SHA-256 digest to
|
||||||
|
match the current resolved stage semantics. Missing legacy evidence, malformed
|
||||||
|
evidence, or a mismatch makes the stage and its fixed transitive dependents
|
||||||
|
stale. The invocation skip copies the matched fingerprint for provenance but
|
||||||
|
does not rewrite session authority. The existing stage-specific resume
|
||||||
|
validator runs only after this semantic check succeeds; both checks are
|
||||||
|
required. Extraction and analyze have resume validators and may reject an
|
||||||
|
otherwise eligible skip when their selected durable evidence is obsolete; the
|
||||||
|
runner marks the aggregate record stale and executes it. Analyze's validator
|
||||||
|
can still accept a partial selection when only unrelated artifact records are
|
||||||
|
stale.
|
||||||
|
|
||||||
|
Implemented reuse coverage is deliberately split between aggregate semantic
|
||||||
|
evidence and focused durable validators:
|
||||||
|
|
||||||
|
| Work | Reuse authority | Focused owners |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| prepare | aggregate semantic fingerprint | [prepare](stage-prepare.md) |
|
||||||
|
| transcribe | aggregate semantic fingerprint | [transcribe](stage-transcribe.md), [WhisperX](../integrations/whisperx.md) |
|
||||||
|
| merge | aggregate semantic fingerprint | [merge](stage-merge.md), [Seriatim](../integrations/seriatim.md) |
|
||||||
|
| polish | aggregate semantic fingerprint | [polish](stage-polish.md), [Audita](../integrations/audita.md) |
|
||||||
|
| normalize | aggregate semantic fingerprint | [normalize](stage-normalize.md), [Seriatim](../integrations/seriatim.md) |
|
||||||
|
| trim | aggregate semantic fingerprint | [trim](stage-trim.md), [Scriptorium](../integrations/scriptorium.md), [Seriatim](../integrations/seriatim.md) |
|
||||||
|
| render | aggregate semantic fingerprint | [render](stage-render.md), [Seriatim](../integrations/seriatim.md) |
|
||||||
|
| extract | aggregate semantic fingerprint plus reference/output validator | [extract](stage-extract.md), [Notarius](../integrations/notarius.md) |
|
||||||
|
| analyze artifacts | per-artifact fingerprint, reconciliation, and output validator | [analyze](stage-analyze.md), [Scriptorium](../integrations/scriptorium.md) |
|
||||||
|
| publish | aggregate semantic fingerprint plus immediate lock/commit checks | [publish](stage-publish.md), [storage adapter](adapters.md) |
|
||||||
|
| notify | aggregate delivery-mode fingerprint | [pipeline overview](overview.md), [configuration](../config.md#notifications) |
|
||||||
|
|
||||||
|
These contracts record resolved choices Narratio can observe, not operational
|
||||||
|
runner tuning. External model, module, prompt, profile, and configuration-file
|
||||||
|
contents that a tool privately loads remain outside the contract when their
|
||||||
|
configured identifier is unchanged; operators must force the affected work
|
||||||
|
after such a private content change.
|
||||||
|
|
||||||
|
Session manifest is the authoritative stage-progress ledger across invocations.
|
||||||
|
Run manifest is invocation-scoped audit state.
|
||||||
|
|
||||||
|
Both manifests retain the most recently resolved invocation's bounded
|
||||||
|
configuration provenance. It identifies the selected profile name and source
|
||||||
|
(`default` or `cli`) plus the effective configuration digest, but never a raw
|
||||||
|
secret or profile content. This provenance is informational: it does not
|
||||||
|
participate in stage resume or cache decisions. A profile change therefore
|
||||||
|
invalidates only stages whose semantic configuration changed. When a private
|
||||||
|
external-tool model, module, prompt, or profile changes behind an unchanged
|
||||||
|
configured identifier, use `--force` for the affected work.
|
||||||
|
|
||||||
|
`session plan` computes the same current fingerprint and applies the same
|
||||||
|
comparison and invalidation rules to a cloned manifest. It predicts the runner
|
||||||
|
decision without persisting session or invocation state. The shared helper
|
||||||
|
hashes deterministic JSON from stage-owned typed structs; stage providers must
|
||||||
|
exclude secrets, complete effective-configuration dumps, and operational
|
||||||
|
values that cannot affect canonical results. Concrete coverage is owned by the
|
||||||
|
focused stage and integration documents linked above.
|
||||||
|
|
||||||
|
Before an explicitly bounded execution starts after `prepare`, the application
|
||||||
|
reads the session manifest and accepts only `succeeded` or `skipped` for every
|
||||||
|
excluded canonical prefix stage. The first other status or absent record fails
|
||||||
|
the request before layout mutation, adapter initialization, session-manifest
|
||||||
|
writes, or run-manifest creation. Excluded prefix records are not passed to
|
||||||
|
resume validators. Records after the selected end are not prerequisites and
|
||||||
|
may be made stale by selected work without being scheduled.
|
||||||
|
|
||||||
|
After a publish commits remotely, any configured local cleanup is first recorded
|
||||||
|
as a session-manifest obligation before deletion begins. Each target becomes
|
||||||
|
complete only after its confined deletion (or safe absence check) and a
|
||||||
|
successful manifest save. An incomplete obligation is retried when publish
|
||||||
|
executes again and retains the committed run and remote identity that authorized
|
||||||
|
it; an invocation that does not execute publish does not perform cleanup.
|
||||||
|
|
||||||
|
Each invocation derives campaign, session, run, local-path, and remote-prefix
|
||||||
|
metadata from the validated resolved configuration as one projection. A persisted
|
||||||
|
session manifest must agree on campaign and session identity before execution;
|
||||||
|
the current projection is refreshed for every invocation while stage progress,
|
||||||
|
inputs, and durable artifacts remain session history.
|
||||||
|
|
||||||
|
For handled failures after an invocation record is created, the runner records
|
||||||
|
the failure on the session ledger and persists it before persisting the failed
|
||||||
|
run audit record. This preserves the resume authority while making a partial
|
||||||
|
persistence disagreement visible. Abrupt process death remains an accepted case
|
||||||
|
where a durable running record can require operator interpretation.
|
||||||
|
|
||||||
|
## Invariants
|
||||||
|
|
||||||
|
- stage resume/skip decisions are session-manifest driven.
|
||||||
|
- semantic fingerprint comparison precedes stage-specific resume validation.
|
||||||
|
- only successful and intentional-skipped results promote current semantic
|
||||||
|
evidence; invocation reuse copies evidence without replacing session state.
|
||||||
|
- running, failed, and self-skipped stages do not retain result payloads from
|
||||||
|
an earlier success.
|
||||||
|
- stale stages retain prior details until replacement execution starts.
|
||||||
|
- force reruns stale succeeded stages in the fixed dependency relation.
|
||||||
|
- run manifest does not replace session manifest as progress authority.
|
||||||
|
- remote commitment is established by a verified current pointer and remote
|
||||||
|
commit relationship, never by a mutable session-manifest boolean.
|
||||||
|
|
||||||
|
## Implementation And Tests
|
||||||
|
|
||||||
|
- Models and transitions: `internal/manifest/manifest.go`,
|
||||||
|
`internal/manifest/run_manifest.go`
|
||||||
|
- Remote commit model and readers: `internal/artifacts/remote_commit.go`,
|
||||||
|
`internal/artifacts/current_state_commit.go`,
|
||||||
|
`internal/artifacts/current_state_legacy.go`
|
||||||
|
- Persistence and validation: `internal/manifest/store.go`
|
||||||
|
- Package tests: `internal/manifest/*_test.go`
|
||||||
|
- Assembled execution behavior: `internal/app/runner_test.go`,
|
||||||
|
`internal/app/run_stage_test.go`
|
||||||
|
|||||||
118
docs/internal/overview.md
Normal file
118
docs/internal/overview.md
Normal file
@@ -0,0 +1,118 @@
|
|||||||
|
# Internal Overview
|
||||||
|
|
||||||
|
This document is the implemented component map for Narratio. Normative system
|
||||||
|
boundaries and dependency direction belong in
|
||||||
|
[Architecture](../policy/architecture.md). User and operator contracts belong
|
||||||
|
in the [CLI](../cli.md), [Configuration](../config.md),
|
||||||
|
[Operations](../operations.md), and [Troubleshooting](../troubleshooting.md).
|
||||||
|
Externally observable tool and format contracts belong under
|
||||||
|
[Integrations](../integrations/).
|
||||||
|
|
||||||
|
## Execution Path
|
||||||
|
|
||||||
|
```text
|
||||||
|
cmd/narratio -> internal/app -> configuration and production composition
|
||||||
|
-> internal/stage -> adapters and external systems
|
||||||
|
-> manifests and artifact resolution -> durable local/remote output
|
||||||
|
```
|
||||||
|
|
||||||
|
The executable delegates process behavior to the application boundary. The
|
||||||
|
application resolves configuration, composes concrete collaborators, acquires
|
||||||
|
session safety controls, and runs commands. Pipeline commands execute the
|
||||||
|
canonical stage sequence through adapter interfaces, while manifests record
|
||||||
|
progress and artifact services resolve durable inputs and outputs.
|
||||||
|
|
||||||
|
## Components
|
||||||
|
|
||||||
|
| Area | Implemented owners | Responsibility |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| Executable | `cmd/narratio` | Process entry, standard stream wiring, argument handoff, and exit status. |
|
||||||
|
| Application orchestration | `internal/app` | Command dispatch, configuration selection, secret-file environment loading, production composition, session locking, planning, execution, restore, cleanup gates, and user-facing reporting. |
|
||||||
|
| Configuration | [`internal/config`](configuration.md) | Presence-aware root/import/profile composition, canonical party and family expansion, strict YAML loading, defaults, normalization, session templating, and validation. |
|
||||||
|
| Pipeline stages | `internal/stage` | Canonical stage registry, shared stage contract, execution dependencies, and implemented stage behavior. |
|
||||||
|
| External boundaries | `internal/adapters`, `internal/audio` | WhisperX HTTP, downstream subprocesses, notification, object storage, and S3 audio materialization behind Narratio contracts. |
|
||||||
|
| Manifests | `internal/manifest` | Durable session progress, invocation audit state, stage transitions, validation, and atomic persistence. |
|
||||||
|
| Artifacts and paths | `internal/artifacts`, `internal/pathsafe` | Artifact identities and resolution, local and remote path/key models, current-state discovery, and confined relative destinations. |
|
||||||
|
| Previous-session cache | `internal/previouscache` | Deterministic planning and materialization requirements for configured previous-session inputs. |
|
||||||
|
| Artifact policy | `internal/artifactpolicy` | Source and destination policy, configured artifact identity validation, and publish destination safety. |
|
||||||
|
| Shared models and file operations | `internal/artifactmodel`, `internal/contracts`, [`internal/fileops`](fileops.md) | Transcript and artifact data contracts plus durable single-file replacement helpers; unsupported directory syncing is reported explicitly. |
|
||||||
|
| Logging | `internal/logging` | Application logger construction and shared structured logging behavior. |
|
||||||
|
|
||||||
|
The application boundary composes concrete implementations. Stages depend on
|
||||||
|
Narratio-level contracts; external transport and SDK details remain in
|
||||||
|
adapters. The normative rules for these relationships remain in
|
||||||
|
[Architecture](../policy/architecture.md).
|
||||||
|
|
||||||
|
Pipeline execution and `session plan` share the same inclusive contiguous-range
|
||||||
|
model. Planning clones session state and applies selected-stage transitions and
|
||||||
|
resume validation in memory; it does not create invocation state or initialize
|
||||||
|
stage-execution adapters. Command configuration loading can still retrieve a
|
||||||
|
missing session file through configured remote storage. It retains the initially
|
||||||
|
composed pipeline and selected campaign while resolving either a local or
|
||||||
|
downloaded remote session, so one invocation cannot mix pipeline revisions.
|
||||||
|
Analyze planning additionally exposes the artifact closure's targets,
|
||||||
|
prerequisite rebuilds, execution order, and current reuse.
|
||||||
|
|
||||||
|
## Pipeline Stage Set
|
||||||
|
|
||||||
|
The implemented canonical order is:
|
||||||
|
|
||||||
|
1. [`prepare`](stage-prepare.md)
|
||||||
|
2. [`transcribe`](stage-transcribe.md)
|
||||||
|
3. [`merge`](stage-merge.md)
|
||||||
|
4. [`polish`](stage-polish.md)
|
||||||
|
5. [`normalize`](stage-normalize.md)
|
||||||
|
6. [`trim`](stage-trim.md)
|
||||||
|
7. [`render`](stage-render.md)
|
||||||
|
8. [`extract`](stage-extract.md)
|
||||||
|
9. [`analyze`](stage-analyze.md)
|
||||||
|
10. [`publish`](stage-publish.md)
|
||||||
|
11. `notify` (no-op)
|
||||||
|
|
||||||
|
`notify` currently has no persisted pipeline outputs and uses the explicit
|
||||||
|
`noop` notification mode. Its versioned semantic evidence records that delivery
|
||||||
|
mode and excludes adapter credentials and response data. The focused stage
|
||||||
|
documents own implementation mechanics. The
|
||||||
|
[CLI](../cli.md) and [Operations](../operations.md) own user-visible invocation
|
||||||
|
and execution semantics.
|
||||||
|
|
||||||
|
Execution order and invalidation are separate application contracts. The stage
|
||||||
|
registry owns the flat execution sequence. The application orchestration owner
|
||||||
|
uses a fixed, validated dependency relation to find transitive dependents in
|
||||||
|
canonical order. In particular, `render` and `extract` are sibling consumers of
|
||||||
|
trimmed transcript state: neither invalidates the other, while either can stale
|
||||||
|
`analyze`, `publish`, and `notify`.
|
||||||
|
|
||||||
|
## Focused Documentation
|
||||||
|
|
||||||
|
- [Configuration Internals](configuration.md): pipeline composition, import
|
||||||
|
confinement, field ownership, decoding, and root-relative path semantics.
|
||||||
|
- [Adapter Internals](adapters.md): external adapter boundaries, composition,
|
||||||
|
failure behavior, and test surfaces.
|
||||||
|
- [Artifact Internals](artifacts.md): source identities, runtime catalog,
|
||||||
|
resolution, previous requirements, and current-state helpers.
|
||||||
|
- [Manifest Internals](manifest.md): session and run records, persistence, and
|
||||||
|
execution transitions.
|
||||||
|
- [Storage Internals](storage.md): object-store interface and S3 behavior.
|
||||||
|
- [Workspace Internals](workspace.md): local layout, locking, and cleanup
|
||||||
|
guardrails.
|
||||||
|
- [Restore Internals](command-restore.md): discovery, planning, execution, and
|
||||||
|
reporting.
|
||||||
|
- [`prepare`](stage-prepare.md)
|
||||||
|
- [`transcribe`](stage-transcribe.md)
|
||||||
|
- [`merge`](stage-merge.md)
|
||||||
|
- [`polish`](stage-polish.md)
|
||||||
|
- [`normalize`](stage-normalize.md)
|
||||||
|
- [`trim`](stage-trim.md)
|
||||||
|
- [`render`](stage-render.md)
|
||||||
|
- [`extract`](stage-extract.md)
|
||||||
|
- [`analyze`](stage-analyze.md)
|
||||||
|
- [`publish`](stage-publish.md)
|
||||||
|
|
||||||
|
Use this map to find an owner, then read its focused documentation and tests
|
||||||
|
before changing behavior.
|
||||||
|
|
||||||
|
The stage registry is implemented in `internal/stage/placeholders.go` and its
|
||||||
|
ordering is protected by `internal/app/planner_test.go`. Cross-invocation skip,
|
||||||
|
force, failure, and invalidation behavior is exercised in
|
||||||
|
`internal/app/runner_test.go` and `internal/app/run_stage_test.go`.
|
||||||
@@ -1,84 +1,185 @@
|
|||||||
# Stage: analyze
|
# Stage: analyze
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Execute selected configured Scriptorium artifacts in deterministic dependency order and promote successful outputs to canonical session artifact paths.
|
|
||||||
|
|
||||||
## Inputs and Outputs
|
Reconcile configured Scriptorium artifacts, execute only required work in
|
||||||
Inputs:
|
dependency order, and safely materialize validated outputs.
|
||||||
- configured artifact definitions from `pipeline.scriptorium.artifacts`
|
|
||||||
- selected artifact filter from runtime (`--artifacts`) when provided
|
|
||||||
- resolved artifact input sources declared per artifact (`inputs.*.source`)
|
|
||||||
- optional previous-session file inputs (`previous_session_artifact`)
|
|
||||||
|
|
||||||
Outputs:
|
## Inputs
|
||||||
- one promoted output file per executed configured artifact at that artifact's configured `output_path`
|
|
||||||
- stage metadata containing generated artifact entries and reused disabled-artifact entries
|
|
||||||
|
|
||||||
## Boundaries
|
- ordinary configured artifacts from `pipeline.scriptorium.artifacts`; canonical
|
||||||
Owns:
|
party artifact families have already expanded into this map during
|
||||||
- runtime artifact catalog construction for analyze execution
|
configuration resolution, including corresponding member dependencies and
|
||||||
- selected-artifact planning and dependency ordering
|
rewritten member-artifact input sources
|
||||||
- per-artifact input resolution, var resolution, timeout/render-debug resolution
|
- optional selected artifact keys supplied through the stage environment
|
||||||
- Scriptorium run/render invocation for each selected artifact
|
- built-in, configured, extraction, and previous-session source references in
|
||||||
- run-local output generation and canonical promotion
|
artifact inputs
|
||||||
|
|
||||||
Does not own:
|
Supported source families:
|
||||||
- transcript generation/processing stages
|
- built-ins: `narratio.transcript.*`, `narratio.bounds.session`
|
||||||
- archive promotion policy
|
- prepared stable inputs: `narratio.input.players`, `narratio.input.party`,
|
||||||
- per-artifact resume semantics
|
`narratio.input.glossary`, `narratio.input.spell_catalog`
|
||||||
|
- configured artifacts: `narratio.artifact.<key>`
|
||||||
|
- extraction lanes: `narratio.extraction.<key>`
|
||||||
|
- previous-session cache: `narratio.previous_session.artifact.<key>`
|
||||||
|
|
||||||
## Config Fields Used
|
## Outputs
|
||||||
- `session.session_id`
|
|
||||||
- `session.campaign`
|
|
||||||
- `pipeline.workspace.root`
|
|
||||||
- `pipeline.scriptorium.binary`
|
|
||||||
- `pipeline.scriptorium.config_path`
|
|
||||||
- `pipeline.scriptorium.timeout`
|
|
||||||
- `pipeline.scriptorium.render_debug`
|
|
||||||
- `pipeline.scriptorium.artifacts.<name>.*`
|
|
||||||
- `enabled`
|
|
||||||
- `depends_on`
|
|
||||||
- `prompt_id`
|
|
||||||
- `profile_id`
|
|
||||||
- `timeout`
|
|
||||||
- `output_path`
|
|
||||||
- `render_debug`
|
|
||||||
- `inputs`
|
|
||||||
- `vars`
|
|
||||||
|
|
||||||
## External Adapters Used
|
- one current per-artifact manifest record per validated materialized output
|
||||||
- Scriptorium adapter:
|
- stage metadata describing selected/generated/reused artifacts
|
||||||
- optional `RenderArtifact` (render debug)
|
|
||||||
- `RunArtifact` (artifact generation)
|
|
||||||
|
|
||||||
## State and Manifest Behavior
|
## Key Behavior
|
||||||
- If `pipeline.scriptorium` is absent, stage returns success metadata with `skipped=true`.
|
|
||||||
- If no artifacts are configured, stage returns success metadata with `skipped=true`.
|
|
||||||
- If zero artifacts are executable after `enabled` + `--artifacts` filtering, stage returns success metadata with `skipped=true`.
|
|
||||||
- Builds runtime catalog with built-ins and configured artifacts.
|
|
||||||
- Non-executable configured artifacts are marked available only when their configured output file exists and is valid on disk.
|
|
||||||
- Executes selected configured artifacts in topological order with deterministic tie-breaking.
|
|
||||||
- For each generated artifact, records metadata fields including `name`, `source_id`, `output_kind`, `path`, `prompt_id`, `profile_id`, and `provenance`.
|
|
||||||
- Reused disabled artifacts are recorded separately in `reused_artifacts` with provenance `filesystem.disabled_artifact_output`.
|
|
||||||
|
|
||||||
## Skip and Resume Behavior
|
- when `pipeline.scriptorium` is absent or no configured artifact is
|
||||||
- Runner-level skip applies when analyze is already `succeeded` and `--force` is not set.
|
executable, completes successfully with no outputs and records explanatory
|
||||||
- Analyze remains stage-scoped for resume/skip; there is no per-artifact resume state.
|
metadata. This is not an explicit self-skip: both manifests record success,
|
||||||
- `--artifacts` filters which configured artifacts are executable when analyze runs; it does not imply `--force`.
|
satisfy publish's prerequisite, and an ordinary later run reuses the result
|
||||||
|
while the effective set remains empty. Enabling or selecting an artifact
|
||||||
|
later makes missing versioned evidence non-resumable and schedules it without
|
||||||
|
requiring force.
|
||||||
|
- builds a runtime artifact catalog containing built-ins, configured artifacts,
|
||||||
|
and configured extraction lanes. Extraction availability is hydrated only
|
||||||
|
from compatible successful extraction evidence.
|
||||||
|
- uses enabled configured artifacts by default. An explicit `--artifacts`
|
||||||
|
selection is a one-invocation override that makes exactly the named
|
||||||
|
configured artifacts explicit targets even when disabled. The work planner
|
||||||
|
adds required configured prerequisites, reuses current ones, and schedules
|
||||||
|
stale, missing, or otherwise non-current prerequisites before dependents.
|
||||||
|
- makes a non-executable configured artifact reusable only when its current
|
||||||
|
manifest record and durable output pass the configured-artifact evidence
|
||||||
|
contract; an incidental or stale canonical file is unavailable.
|
||||||
|
- validates selected artifact dependency order (cycle-safe topo ordering).
|
||||||
|
- resolves required/optional inputs per artifact source definition into an
|
||||||
|
ordered semantic identity. Each identity records the configured input name,
|
||||||
|
canonical source ID, required policy, explicit presence, source contract,
|
||||||
|
checksum, size, and a source-based logical identity. Workspace paths and
|
||||||
|
producer run IDs are excluded.
|
||||||
|
- orders input identities by configured input name independently of Go map
|
||||||
|
iteration. Runtime adapter paths remain a separate execution-only map.
|
||||||
|
- omits an unavailable optional input from the adapter request while retaining
|
||||||
|
explicit absence in its semantic identity; an unavailable required input
|
||||||
|
fails.
|
||||||
|
- resolves prepared stable input sources through the shared manifest-authoritative
|
||||||
|
identity resolver; it does not accept incidental files or fall back to
|
||||||
|
campaign/session source paths.
|
||||||
|
- reuses checksums and sizes from validated prepared, extraction, and current
|
||||||
|
configured-artifact evidence. Other resolved inputs are hashed as confined
|
||||||
|
regular files with streaming reads and the central resolved-artifact size
|
||||||
|
limit.
|
||||||
|
- owns a versioned SHA-256 fingerprint contract with one fixed-field canonical
|
||||||
|
JSON payload and no map serialization. Configured artifacts are fingerprinted
|
||||||
|
in deterministic dependency order.
|
||||||
|
- fingerprints the normalized artifact key, prompt and profile identifiers,
|
||||||
|
effective render-debug behavior, session-relative output identity, sorted
|
||||||
|
dependency keys, ordered input declarations and semantic identities,
|
||||||
|
validated current dependency-output identities, and sorted effective
|
||||||
|
Scriptorium variables (including Narratio's sticky session variable).
|
||||||
|
- provides read-only reconciliation that classifies each configured record as
|
||||||
|
current, stale, missing, failed, legacy, or otherwise non-resumable, and
|
||||||
|
separately identifies manifest records removed from current configuration.
|
||||||
|
A record is current only when its fingerprint version and value match and its
|
||||||
|
configured output still passes manifest-authoritative evidence validation.
|
||||||
|
- owns a read-only typed work planner. Its explicit targets are enabled
|
||||||
|
artifacts by default or the exact normalized `--artifacts` selection when
|
||||||
|
supplied. It closes targets over configured prerequisites, orders the closure
|
||||||
|
topologically, reuses current members, and schedules every non-current member
|
||||||
|
before its dependents.
|
||||||
|
- force applies only to explicit targets. A current prerequisite is reused
|
||||||
|
unless it is itself an explicit forced target; disabled prerequisites may be
|
||||||
|
rebuilt when required, while unrelated disabled artifacts are excluded.
|
||||||
|
- the work plan carries explicit targets, prerequisite-only work, deterministic
|
||||||
|
execution and reuse lists, invalidated and removed records, and a cloned
|
||||||
|
projected record collection. Valid unrelated configured records survive the
|
||||||
|
projection, removed records are omitted, and legacy files never become
|
||||||
|
current without regeneration.
|
||||||
|
- implements aggregate resume validation by running the same read-only catalog,
|
||||||
|
fingerprint reconciliation, and work planner used by execution. A succeeded
|
||||||
|
aggregate record is reusable exactly when the selected closure schedules no
|
||||||
|
artifact work; stale unrelated records do not block a partial selection.
|
||||||
|
- exposes the typed artifact decision to `session plan`. Planning applies it to
|
||||||
|
a cloned manifest after modeling earlier selected stage transitions, so
|
||||||
|
aggregate run/skip and artifact execute/reuse decisions match the ordinary
|
||||||
|
runner without creating durable state or invoking Scriptorium.
|
||||||
|
- executes only the work plan's scheduled entries. Manifest-validated current
|
||||||
|
prerequisites remain available through the runtime catalog without invoking
|
||||||
|
Scriptorium; newly produced prerequisites enter that catalog with the same
|
||||||
|
contract, checksum, and size identity used for persisted current evidence.
|
||||||
|
- keeps adapter output in the invocation's run-local analyze directory until
|
||||||
|
it is a safe, non-empty, bounded regular file with a calculated checksum and
|
||||||
|
complete output contract. Canonical replacement uses the shared atomic file
|
||||||
|
operation boundary and verifies that the installed checksum matches the
|
||||||
|
validated run-local bytes.
|
||||||
|
- records each successful artifact's freshly computed fingerprint, canonical
|
||||||
|
relative output path, contract, checksum, size, producer run ID, bounded
|
||||||
|
Scriptorium provenance, logs, and generated configuration references in the
|
||||||
|
analyze-owned projection.
|
||||||
|
- preserves valid unrelated current records during partial execution. If a
|
||||||
|
rebuilt output's bytes and contract are unchanged, unselected dependents may
|
||||||
|
remain current. If that semantic identity changes, unselected transitive
|
||||||
|
dependents become stale without being executed; dependents included in the
|
||||||
|
invocation are evaluated in dependency order instead.
|
||||||
|
- reports all evaluated targets and prerequisites in invocation state. The
|
||||||
|
runner reconstructs aggregate session outputs from every current session
|
||||||
|
record and invocation outputs from only records produced by the current run.
|
||||||
|
Unrelated stale records do not make an otherwise successful partial
|
||||||
|
invocation fail.
|
||||||
|
- resolves previous-session sources from local `previous/` cache only.
|
||||||
|
- runs optional render-debug, then artifact execution.
|
||||||
|
- validates non-empty output files and materializes canonical outputs.
|
||||||
|
|
||||||
## Failure Behavior
|
## Failure Semantics
|
||||||
- Fails on invalid dependency ordering, unavailable required configured inputs, invalid built-in input prerequisites, render/run adapter failures, validation-failed adapter results, or missing/empty outputs.
|
|
||||||
- Required configured dependency missing from catalog availability fails clearly before invocation.
|
|
||||||
- Optional missing inputs are omitted.
|
|
||||||
|
|
||||||
## Tests to Inspect Before Changing
|
- required missing configured/previous-session inputs fail.
|
||||||
- `internal/stage/analyze_test.go`
|
- missing required prepared stable input source includes prepare rerun guidance.
|
||||||
- `internal/artifacts/catalog_test.go`
|
- missing required previous-session source includes prepare rerun guidance.
|
||||||
- `internal/artifacts/artifact_resolver_test.go`
|
- missing required `narratio.transcript.final_markdown` or
|
||||||
- `internal/adapters/scriptorium/subprocess_test.go`
|
`narratio.transcript.final_trimmed_markdown` inputs includes render rerun
|
||||||
|
guidance.
|
||||||
|
- dependency cycles or unavailable required dependencies fail.
|
||||||
|
- adapter validation failures fail stage.
|
||||||
|
- a scheduled artifact failure returns the restricted analyze-state projection
|
||||||
|
with the active artifact marked `failed`, a bounded error, and no output
|
||||||
|
authority. Current transitive dependents become stale without execution.
|
||||||
|
- earlier artifacts from the invocation remain current only after their
|
||||||
|
run-local output passed validation and canonical materialization. They remain
|
||||||
|
in invocation history; unattempted later artifacts do not appear there.
|
||||||
|
- unrelated current records survive a partial failure. Old canonical bytes for
|
||||||
|
the failed artifact and newly materialized bytes whose projection cannot be
|
||||||
|
persisted are incidental, not current evidence.
|
||||||
|
- the runner persists a valid partial projection before it marks aggregate
|
||||||
|
analyze failed and invalidates publish and notify through the application
|
||||||
|
dependency relation. Projection-persistence errors retain the last durable
|
||||||
|
per-artifact authority and are joined with the original failure context.
|
||||||
|
|
||||||
## Architectural Invariants
|
## Invariants
|
||||||
- Configured artifacts are identified by `narratio.artifact.<name>` source IDs.
|
|
||||||
- Artifact-to-artifact references rely on explicit `depends_on` declarations validated in config.
|
- `analyze` performs no remote storage calls for previous-session source resolution.
|
||||||
- Generated analyze outputs are treated uniformly as Scriptorium artifacts.
|
- input-identity resolution is read-only: it does not invoke adapters,
|
||||||
- Successful outputs must exist and be non-empty before promotion.
|
materialize outputs, update status, or create run records.
|
||||||
|
- fingerprints exclude timeouts, retries, timestamps, producer and Narratio run
|
||||||
|
IDs, executable and config paths, workspace roots, diagnostic locations, and
|
||||||
|
executable or private transitive configuration contents. A change that is
|
||||||
|
visible only inside Scriptorium—such as a file privately loaded by its config
|
||||||
|
path—requires an explicit forced regeneration.
|
||||||
|
- output provenance and metadata are deterministic per execution.
|
||||||
|
- a canonical file without current per-artifact manifest evidence is never
|
||||||
|
promoted to current state.
|
||||||
|
|
||||||
|
## Related Contracts And Tests
|
||||||
|
|
||||||
|
- [Configuration](../config.md#scriptorium-artifact-entries) owns artifact
|
||||||
|
fields and source-selection rules, including
|
||||||
|
[artifact families](../config.md#scriptorium-artifact-families).
|
||||||
|
- [CLI](../cli.md) owns user-visible artifact selection.
|
||||||
|
- [Scriptorium](../integrations/scriptorium.md) owns the subprocess contract.
|
||||||
|
- Implementation and tests: `internal/stage/analyze.go`,
|
||||||
|
`internal/stage/analyze_input_identity.go`, `internal/stage/analyze_test.go`,
|
||||||
|
`internal/stage/analyze_input_identity_test.go`,
|
||||||
|
`internal/stage/analyze_fingerprint.go`,
|
||||||
|
`internal/stage/analyze_fingerprint_test.go`,
|
||||||
|
`internal/stage/analyze_reconciliation.go`, and
|
||||||
|
`internal/stage/analyze_reconciliation_test.go`,
|
||||||
|
`internal/stage/analyze_plan.go`, `internal/stage/analyze_plan_test.go`, and
|
||||||
|
`internal/stage/analyze_incremental_execution_test.go`, and
|
||||||
|
`internal/stage/analyze_failure_test.go`,
|
||||||
|
`internal/stage/analyze_resume.go`, and `internal/stage/analyze_resume_test.go`
|
||||||
|
|||||||
@@ -1,68 +0,0 @@
|
|||||||
# Stage: archive
|
|
||||||
|
|
||||||
## Purpose
|
|
||||||
Publish run records and promoted session artifacts to object storage, then atomically advance the remote current pointer.
|
|
||||||
|
|
||||||
## Inputs and Outputs
|
|
||||||
Inputs:
|
|
||||||
- session manifest and prerequisite stage records
|
|
||||||
- run root contents under `runs/{run_id}/`
|
|
||||||
- promotion sources from session root (`archive.promote_artifacts`)
|
|
||||||
|
|
||||||
Outputs:
|
|
||||||
- uploaded run files under `{session_prefix}/runs/{run_id}/...`
|
|
||||||
- uploaded promoted artifacts under `{session_prefix}/...`
|
|
||||||
- `{session_prefix}/current/manifest.json`
|
|
||||||
- `{session_prefix}/current/run_id.txt` written last
|
|
||||||
|
|
||||||
## Boundaries
|
|
||||||
Owns:
|
|
||||||
- Archive enable/disable gate behavior
|
|
||||||
- Prerequisite stage success enforcement
|
|
||||||
- Run file collection and upload (excluding `audio/`)
|
|
||||||
- Promotion rule resolution and upload
|
|
||||||
- Commit pointer publish order
|
|
||||||
|
|
||||||
Does not own:
|
|
||||||
- Stage execution before archive
|
|
||||||
- Post-archive local cleanup policy execution (handled by app cleanup logic)
|
|
||||||
|
|
||||||
## Config Fields Used
|
|
||||||
- `pipeline.archive.enabled`
|
|
||||||
- `pipeline.archive.upload_run`
|
|
||||||
- `pipeline.archive.promote_artifacts`
|
|
||||||
- `pipeline.storage.s3.bucket`
|
|
||||||
- `pipeline.storage.s3.root_prefix`
|
|
||||||
- `pipeline.workspace.root`
|
|
||||||
- `session.campaign`
|
|
||||||
- `session.session_id`
|
|
||||||
|
|
||||||
## External Adapters Used
|
|
||||||
- Object storage backend (`env.ObjectStore`) for upload/list primitives.
|
|
||||||
|
|
||||||
## State and Manifest Behavior
|
|
||||||
- Requires `prepare`, `transcribe`, `merge`, `polish`, `normalize`, `trim`, and `analyze` status `succeeded`.
|
|
||||||
- Resolves bucket/prefix from manifest identity first, then config fallback.
|
|
||||||
- Writes metadata including:
|
|
||||||
- upload counts/paths
|
|
||||||
- `current_manifest_key`
|
|
||||||
- `current_run_id_key`
|
|
||||||
- `current_pointer_written`
|
|
||||||
- On skipped archive path, returns metadata with `skipped=true` and pointer not written.
|
|
||||||
|
|
||||||
## Skip and Resume Behavior
|
|
||||||
- Stage may self-skip (metadata skip) when archive disabled or run upload disabled.
|
|
||||||
- Runner-level skip also applies for previously succeeded stage unless forced.
|
|
||||||
|
|
||||||
## Failure Behavior
|
|
||||||
- Fails on missing prerequisite success, missing object store when required, missing run root, missing required promotion source, upload failures, or pointer write failures.
|
|
||||||
- Pointer semantics are fail-safe: `current/run_id.txt` is not written if prior required uploads fail.
|
|
||||||
|
|
||||||
## Tests to Inspect Before Changing
|
|
||||||
- `internal/stage/archive_test.go`
|
|
||||||
- `internal/app/post_archive_cleanup_test.go`
|
|
||||||
|
|
||||||
## Architectural Invariants
|
|
||||||
- Run upload excludes `audio/` subtree.
|
|
||||||
- `current/manifest.json` uploads before `current/run_id.txt`.
|
|
||||||
- `current/run_id.txt` is the remote publish commit marker.
|
|
||||||
118
docs/internal/stage-extract.md
Normal file
118
docs/internal/stage-extract.md
Normal file
@@ -0,0 +1,118 @@
|
|||||||
|
# Internal: Extract Stage
|
||||||
|
|
||||||
|
## Responsibility
|
||||||
|
|
||||||
|
`extract` runs after `render` and before `analyze`. It converts the canonical
|
||||||
|
`narratio.transcript.final_trimmed` JSON into configured Notarius lane artifacts.
|
||||||
|
An omitted or disabled Notarius section makes the stage explicitly self-skip
|
||||||
|
with reason `notarius_disabled`, no outputs, and no Notarius runner.
|
||||||
|
|
||||||
|
The external protocol is documented in the
|
||||||
|
[Notarius integration contract](../integrations/notarius.md). Configuration
|
||||||
|
fields belong in [Configuration](../config.md), and physical paths and force
|
||||||
|
procedures belong in [Operations](../operations.md).
|
||||||
|
|
||||||
|
## Lifecycle
|
||||||
|
|
||||||
|
`internal/stage/extract.go`:
|
||||||
|
|
||||||
|
1. resolves the final trimmed transcript from the shared artifact catalog;
|
||||||
|
2. resolves every configured prepared reference through the shared
|
||||||
|
manifest-authoritative identity resolver before creating run-local output;
|
||||||
|
3. streams each verified reference into an invocation-local snapshot and
|
||||||
|
rejects any source change observed while copying;
|
||||||
|
4. fingerprints the byte- and provenance-bearing Notarius invocation evidence,
|
||||||
|
including sorted reference identities;
|
||||||
|
5. creates a run-local staging directory and invokes the injected
|
||||||
|
`notarius.Runner`;
|
||||||
|
6. revalidates the reference snapshots, then validates the v2 successful
|
||||||
|
receipt, confined index, management documents, configured required lane
|
||||||
|
descriptors, validation summaries, and regular payload files;
|
||||||
|
7. atomically promotes the complete bundle to its immutable durable location;
|
||||||
|
8. records one non-selectable `notarius_index` output and one selectable
|
||||||
|
`notarius_lane` output per configured lane; and
|
||||||
|
9. registers each lane as `narratio.extraction.<output_key>` for downstream
|
||||||
|
Scriptorium and publish resolution.
|
||||||
|
|
||||||
|
Lane records retain checksum, contract, producer run ID, and Notarius system,
|
||||||
|
run, pipeline, and lane provenance. Stage metadata retains the durable bundle
|
||||||
|
root, receipt, diagnostic paths, rejection/warning summaries, producing
|
||||||
|
Narratio run ID, the resolved trimmed-input identity, and invocation
|
||||||
|
fingerprint. The input identity binds the exact transcript bytes, canonical
|
||||||
|
source ID, producer stage/output/run identity, and resolution provenance.
|
||||||
|
Reference metadata contains only selector, source ID, canonical session-relative
|
||||||
|
path, checksum, and size; adapter requests receive selector and absolute
|
||||||
|
invocation-local snapshot path, never payload contents. Snapshot bytes must
|
||||||
|
match the prepared identity both before and after Notarius runs, so a concurrent
|
||||||
|
prepared-file replacement cannot make recorded provenance describe different
|
||||||
|
bytes from those supplied to Notarius.
|
||||||
|
Validation completes before
|
||||||
|
promotion, so a rejected result cannot expose a partial durable bundle.
|
||||||
|
|
||||||
|
Any executed extraction outcome that replaces a different effective outcome
|
||||||
|
marks succeeded analysis and delivery dependents stale. Render is an independent
|
||||||
|
sibling and remains current. Repeating the same disabled self-skip with no
|
||||||
|
outputs is stable and does not repeatedly invalidate dependent stages.
|
||||||
|
|
||||||
|
## Resume Validation
|
||||||
|
|
||||||
|
Before the focused validator runs, the application compares extract's versioned
|
||||||
|
semantic fingerprint. It covers enablement, Notarius pipeline identity, sorted
|
||||||
|
reference selector/source mappings, sorted declared output contracts, and each
|
||||||
|
canonical `narratio.extraction.<key>` output identity. It excludes executable,
|
||||||
|
timeout, working directory, config path, and private Notarius config contents.
|
||||||
|
|
||||||
|
`internal/stage/extract_resume.go` then permits a skip only when the existing
|
||||||
|
stage record still matches the current byte- and provenance-bearing invocation
|
||||||
|
evidence. That evidence covers the current direct trimmed-transcript identity,
|
||||||
|
sorted prepared-reference identities, pipeline identity, and configured output
|
||||||
|
contracts. The same reference helper and transcript identity are resolved again
|
||||||
|
for artifact evidence, so changing current transcript bytes, reference bytes,
|
||||||
|
or producer identity makes the prior extraction obsolete. Operational runner
|
||||||
|
settings do not invalidate otherwise current durable evidence.
|
||||||
|
|
||||||
|
A valid prepared-reference change makes extraction non-resumable. Missing,
|
||||||
|
unsafe, or checksum-inconsistent prepared evidence is a hard validation error
|
||||||
|
with prepare-force guidance because an immediate extract rerun cannot succeed.
|
||||||
|
|
||||||
|
The validator then checks the producing run identity, canonical immutable
|
||||||
|
bundle root, path confinement and absence of symlink components, receipt
|
||||||
|
identity, exactly one canonical index, the exact configured source set,
|
||||||
|
contracts and provenance, regular-file status, and stored checksums. Missing or
|
||||||
|
obsolete results are non-resumable and run again; unsafe filesystem conditions
|
||||||
|
return an error rather than silently accepting or replacing data.
|
||||||
|
|
||||||
|
Neither contract can observe files imported by Notarius configuration, profile
|
||||||
|
contents, prompt/module definitions, or other transitive inputs. Operators must
|
||||||
|
force extraction after changing any such private input behind a stable
|
||||||
|
identifier.
|
||||||
|
|
||||||
|
## Failure Behavior
|
||||||
|
|
||||||
|
Adapter startup, timeout, nonzero exit, receipt decoding, path confinement,
|
||||||
|
index compatibility, inconsistent warning or diagnostic envelopes,
|
||||||
|
required-lane rejection or incomplete validation, payload inspection,
|
||||||
|
checksum, or promotion errors fail the stage through ordinary manifest
|
||||||
|
transition handling.
|
||||||
|
Stdout receipt and stderr diagnostics remain separate. Downstream stages are
|
||||||
|
not given selectable extraction sources unless the complete configured result
|
||||||
|
has passed validation and promotion.
|
||||||
|
|
||||||
|
When a replacement attempt begins, the current session-stage record no longer
|
||||||
|
advertises payload from the previous success. A failed replacement therefore
|
||||||
|
has no current outputs, logs, generated configuration references, or metadata,
|
||||||
|
while the earlier invocation manifest and immutable promoted bundle remain
|
||||||
|
available for audit and recovery.
|
||||||
|
|
||||||
|
## Implementation And Focused Tests
|
||||||
|
|
||||||
|
- Stage execution, selection, and resume validation: `internal/stage/extract.go`,
|
||||||
|
`internal/stage/extract_resume.go`,
|
||||||
|
`internal/stage/extract_test.go`,
|
||||||
|
`internal/stage/semantic_contracts_delivery.go`
|
||||||
|
- Subprocess boundary: `internal/adapters/notarius/subprocess.go`,
|
||||||
|
`internal/adapters/notarius/subprocess_test.go`
|
||||||
|
- Catalog hydration: `internal/artifacts/extraction_catalog.go`,
|
||||||
|
`internal/artifacts/extraction_catalog_test.go`
|
||||||
|
- Composition and downstream behavior: `internal/app/runner_test.go`,
|
||||||
|
`internal/stage/analyze_test.go`, `internal/stage/publish_test.go`
|
||||||
@@ -1,63 +1,51 @@
|
|||||||
# Stage: merge
|
# Stage: merge
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Normalize per-speaker raw transcripts and merge them into one merged transcript via Seriatim.
|
|
||||||
|
|
||||||
## Inputs and Outputs
|
Normalize raw transcript inputs and merge into base transcript via Seriatim.
|
||||||
Inputs:
|
|
||||||
|
## Inputs
|
||||||
|
|
||||||
- `transcripts/raw/*.json`
|
- `transcripts/raw/*.json`
|
||||||
- `inputs/speakers.yml`
|
- `inputs/speakers.yml`
|
||||||
- `inputs/autocorrect.yml`
|
- `inputs/autocorrect.yml`
|
||||||
|
|
||||||
Outputs:
|
## Outputs
|
||||||
- `transcripts/merged.json`
|
|
||||||
- optional `artifacts/seriatim.report.json` (when report enabled)
|
|
||||||
|
|
||||||
## Boundaries
|
- `transcripts/base.json`
|
||||||
Owns:
|
- optional `artifacts/seriatim.report.json`
|
||||||
- Raw transcript discovery/validation
|
|
||||||
- Per-input normalize calls to Seriatim
|
|
||||||
- Final merge call to Seriatim
|
|
||||||
- Run-local log/config/report path wiring
|
|
||||||
- Promotion of merged/report outputs to canonical paths
|
|
||||||
|
|
||||||
Does not own:
|
## Key Behavior
|
||||||
- Transcript polishing or downstream artifact generation
|
|
||||||
|
|
||||||
## Config Fields Used
|
- discovers and validates raw transcript inputs.
|
||||||
- `session.session_id`
|
- normalizes each raw transcript (`seriatim.Normalize`) into run-local scratch output.
|
||||||
- `session.campaign`
|
- merges normalized inputs (`seriatim.Run`) into base transcript.
|
||||||
- `pipeline.workspace.root`
|
- validates merged transcript and optional report JSON.
|
||||||
- `pipeline.seriatim.binary`
|
- materializes canonical outputs and records stage logs/generated configs.
|
||||||
- `pipeline.seriatim.timeout`
|
|
||||||
- `pipeline.seriatim.output_schema`
|
|
||||||
- `pipeline.seriatim.coalesce_gap`
|
|
||||||
- `pipeline.seriatim.report`
|
|
||||||
- `pipeline.seriatim.env.*`
|
|
||||||
|
|
||||||
## External Adapters Used
|
## Invariants
|
||||||
- Seriatim adapter:
|
|
||||||
- `Normalize` for each raw input
|
|
||||||
- `Run` for final merge
|
|
||||||
|
|
||||||
## State and Manifest Behavior
|
- merge always consumes normalized forms of raw inputs.
|
||||||
- Reads transcript inputs from transcribe stage outputs in manifest when present; falls back to canonical raw directory.
|
- base transcript must validate before stage success.
|
||||||
- Writes run-local outputs/logs/config under `runs/{run_id}/merge/...` when enabled.
|
- report output is config-gated.
|
||||||
- Promotes canonical merged transcript and optional report.
|
|
||||||
- Records normalized-input provenance and adapter metadata in stage metadata.
|
|
||||||
|
|
||||||
## Skip and Resume Behavior
|
## Resume Evidence
|
||||||
- Runner-level skip applies when already succeeded and not forced.
|
|
||||||
- Forced rerun of this or upstream stages can stale downstream succeeded stages via runner invalidation.
|
|
||||||
|
|
||||||
## Failure Behavior
|
Merge records a versioned semantic-configuration fingerprint for the Seriatim
|
||||||
- Fails on missing/invalid raw transcripts, missing speakers/autocorrect files, normalize failure, merge failure, invalid merged output JSON, or invalid report JSON when enabled.
|
merge operation, output schema, coalesce gap, and every configured advanced
|
||||||
|
merge transformation. A change reruns merge and stales only its fixed
|
||||||
|
descendants; prepare and transcribe remain reusable. Binary path, timeout,
|
||||||
|
report emission, logs, and diagnostic retention are operational exclusions.
|
||||||
|
|
||||||
## Tests to Inspect Before Changing
|
Configuration or resources loaded privately inside Seriatim are outside
|
||||||
- `internal/stage/merge_test.go`
|
Narratio's observable contract and require `--force` when changed. An existing
|
||||||
- `internal/adapters/seriatim/subprocess_test.go`
|
successful merge record without evidence reruns once when selected.
|
||||||
|
|
||||||
## Architectural Invariants
|
## Related Contracts And Tests
|
||||||
- Merge consumes normalized forms of each raw transcript.
|
|
||||||
- Merged transcript must validate before promotion.
|
- [Seriatim](../integrations/seriatim.md) owns subprocess and output semantics.
|
||||||
- Report output is optional and gated by config.
|
- [Configuration](../config.md#pipeline) owns operator-selected Seriatim values.
|
||||||
|
- Implementation and tests: `internal/stage/merge.go`,
|
||||||
|
`internal/stage/merge_test.go`,
|
||||||
|
`internal/stage/semantic_contracts_initial.go`, and
|
||||||
|
`internal/stage/semantic_contracts_initial_test.go`
|
||||||
|
|||||||
@@ -1,56 +1,42 @@
|
|||||||
# Stage: normalize
|
# Stage: normalize
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Normalize the processed transcript into a deterministic intermediate schema for trim and optionally emit a normalize report.
|
|
||||||
|
|
||||||
## Inputs and Outputs
|
Normalize polished transcript into final transcript using Seriatim.
|
||||||
Inputs:
|
|
||||||
- `transcripts/processed.json`
|
|
||||||
|
|
||||||
Outputs:
|
## Inputs
|
||||||
- `transcripts/normalized.json` (or configured normalize output path)
|
|
||||||
|
- `transcripts/polished.json`
|
||||||
|
|
||||||
|
## Outputs
|
||||||
|
|
||||||
|
- `transcripts/final.json` (or configured normalize output path)
|
||||||
- optional `artifacts/seriatim.normalize.report.json`
|
- optional `artifacts/seriatim.normalize.report.json`
|
||||||
|
|
||||||
## Boundaries
|
## Key Behavior
|
||||||
Owns:
|
|
||||||
- Processed transcript discovery/validation
|
|
||||||
- Normalize request construction and invocation
|
|
||||||
- Optional normalize report wiring
|
|
||||||
- Promotion of normalized transcript and optional report
|
|
||||||
|
|
||||||
Does not own:
|
- resolves polished transcript from manifest outputs/canonical fallback.
|
||||||
- Bounds detection or segment trimming
|
- applies `pipeline.normalize` config or default normalize config.
|
||||||
|
- runs Seriatim normalize with configured timeout/binary.
|
||||||
|
- validates normalized transcript and optional report.
|
||||||
|
- materializes canonical outputs and records logs/generated configs.
|
||||||
|
|
||||||
## Config Fields Used
|
## Invariants
|
||||||
- `session.session_id`
|
|
||||||
- `session.campaign`
|
|
||||||
- `pipeline.workspace.root`
|
|
||||||
- `pipeline.normalize.output_path`
|
|
||||||
- `pipeline.normalize.output_schema`
|
|
||||||
- `pipeline.normalize.report`
|
|
||||||
- `pipeline.seriatim.binary`
|
|
||||||
- `pipeline.seriatim.timeout`
|
|
||||||
|
|
||||||
## External Adapters Used
|
- final transcript must validate as processed transcript JSON (`segments` array).
|
||||||
- Seriatim adapter (`Normalize`).
|
- normalize defaults are applied when `pipeline.normalize` is unset.
|
||||||
|
|
||||||
## State and Manifest Behavior
|
## Resume Semantics
|
||||||
- Reads processed transcript from polish outputs in manifest when present; falls back to canonical path.
|
|
||||||
- Uses run-local output/report/log/config paths when run layout is enabled.
|
|
||||||
- Promotes canonical normalized transcript and optional normalize report.
|
|
||||||
- Records adapter/result metadata including source path selection.
|
|
||||||
|
|
||||||
## Skip and Resume Behavior
|
The versioned semantic fingerprint covers the Seriatim normalize operation,
|
||||||
- Runner-level skip applies when already succeeded and not forced.
|
output schema and canonical output identity, plus the configured transcript
|
||||||
- Forced reruns can stale downstream succeeded stages.
|
transformations. Seriatim's executable and timeout and optional report
|
||||||
|
generation are operational and do not invalidate the normalized transcript.
|
||||||
|
|
||||||
## Failure Behavior
|
## Related Contracts And Tests
|
||||||
- Fails on missing/invalid processed transcript, adapter error, invalid normalized output, or invalid report output when report enabled.
|
|
||||||
|
|
||||||
## Tests to Inspect Before Changing
|
- [Seriatim](../integrations/seriatim.md) owns subprocess and output semantics.
|
||||||
- `internal/stage/normalize_test.go`
|
- [Configuration](../config.md#pipeline) owns normalize fields and defaults.
|
||||||
- `internal/adapters/seriatim/subprocess_test.go`
|
- Implementation and tests: `internal/stage/normalize.go`,
|
||||||
|
`internal/stage/normalize_test.go`,
|
||||||
## Architectural Invariants
|
`internal/stage/semantic_contracts_refinement.go`
|
||||||
- Normalized output must validate as processed-transcript-compatible JSON (`segments` array required).
|
|
||||||
- Default normalize config is applied when `pipeline.normalize` is unset.
|
|
||||||
|
|||||||
@@ -1,69 +1,53 @@
|
|||||||
# Stage: polish
|
# Stage: polish
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Polish merged transcript with Audita and produce a processed transcript for downstream normalization/analyze.
|
|
||||||
|
|
||||||
## Inputs and Outputs
|
Run Audita polishing on base transcript and produce polished transcript.
|
||||||
Inputs:
|
|
||||||
- `transcripts/merged.json`
|
## Inputs
|
||||||
|
|
||||||
|
- `transcripts/base.json`
|
||||||
- `inputs/glossary.yml`
|
- `inputs/glossary.yml`
|
||||||
|
|
||||||
Outputs:
|
## Outputs
|
||||||
- `transcripts/processed.json`
|
|
||||||
- optional `artifacts/audita.report.json` (when report enabled)
|
|
||||||
|
|
||||||
## Boundaries
|
- `transcripts/polished.json`
|
||||||
Owns:
|
- optional `artifacts/audita.report.json`
|
||||||
- Merged transcript discovery/validation
|
|
||||||
- Audita invocation request construction
|
|
||||||
- Run-local logs/config/work-dir/report wiring
|
|
||||||
- Promotion of processed transcript and optional report
|
|
||||||
|
|
||||||
Does not own:
|
## Key Behavior
|
||||||
- Upstream merge normalization
|
|
||||||
- Downstream normalize/trim/analyze logic
|
|
||||||
|
|
||||||
## Config Fields Used
|
- resolves base transcript from merge outputs/canonical fallback.
|
||||||
- `session.session_id`
|
- invokes an Audita runner configured with static model/runtime options; the
|
||||||
- `session.campaign`
|
invocation supplies paths and modules.
|
||||||
- `pipeline.workspace.root`
|
- validates processed transcript structure (`segments` array required).
|
||||||
- `pipeline.audita.binary`
|
- validates optional report JSON.
|
||||||
- `pipeline.audita.timeout`
|
- materializes canonical outputs; records logs/generated config and adapter metadata.
|
||||||
- `pipeline.audita.llm_api_key_env`
|
|
||||||
- `pipeline.audita.modules`
|
|
||||||
- `pipeline.audita.base_url`
|
|
||||||
- `pipeline.audita.model`
|
|
||||||
- `pipeline.audita.transcript_description`
|
|
||||||
- `pipeline.audita.config_path`
|
|
||||||
- `pipeline.audita.output_schema`
|
|
||||||
- `pipeline.audita.work_dir_retention`
|
|
||||||
- `pipeline.audita.total_llm_concurrency`
|
|
||||||
- `pipeline.audita.proposal_llm_concurrency`
|
|
||||||
- `pipeline.audita.validation_model`
|
|
||||||
- `pipeline.audita.validation_llm_concurrency`
|
|
||||||
- `pipeline.audita.report`
|
|
||||||
|
|
||||||
## External Adapters Used
|
## Invariants
|
||||||
- Audita adapter (`env.Audita.Run`).
|
|
||||||
|
|
||||||
## State and Manifest Behavior
|
- polished transcript schema validation is mandatory.
|
||||||
- Reads merged transcript from merge manifest outputs when available; falls back to canonical merged path.
|
- report output is config-gated.
|
||||||
- Uses run-local output/report/log/config/scratch paths when run layout is enabled.
|
|
||||||
- Promotes canonical `transcripts/processed.json` and optional report.
|
|
||||||
- Records adapter invocation metadata, credential presence signal, and output provenance in stage metadata.
|
|
||||||
|
|
||||||
## Skip and Resume Behavior
|
## Resume Semantics
|
||||||
- Runner-level skip applies when already succeeded and not forced.
|
|
||||||
- Forced rerun can stale downstream succeeded stages via runner invalidation.
|
|
||||||
|
|
||||||
## Failure Behavior
|
The versioned semantic fingerprint covers the Audita service endpoint, model,
|
||||||
- Fails on missing/invalid merged transcript, missing glossary, adapter error, invalid processed output shape (`segments` array required), or invalid report JSON when enabled.
|
validation model, module set, transcript description, output schema, selected
|
||||||
|
external configuration path, and canonical polished-transcript identity. Module
|
||||||
|
ordering is normalized because the configured modules form a set. Audita's
|
||||||
|
executable, timeouts, concurrency, report and debug behavior, work retention,
|
||||||
|
and credential environment name are operational and do not invalidate a
|
||||||
|
successful result.
|
||||||
|
|
||||||
## Tests to Inspect Before Changing
|
Narratio can fingerprint a selected model, module, or configuration identifier,
|
||||||
- `internal/stage/polish_test.go`
|
but it cannot inspect content that Audita privately resolves behind that stable
|
||||||
- `internal/adapters/audita/subprocess_test.go`
|
identifier. Force `polish` after changing such private content without changing
|
||||||
|
its identifier.
|
||||||
|
|
||||||
## Architectural Invariants
|
## Related Contracts And Tests
|
||||||
- Processed transcript must contain a top-level `segments` array.
|
|
||||||
- Report behavior is strictly config-gated.
|
- [Audita](../integrations/audita.md) owns subprocess, validation, and failure
|
||||||
- Stage output canonicalization always ends at `transcripts/processed.json`.
|
semantics.
|
||||||
|
- [Configuration](../config.md#pipeline) owns operator-selected Audita values.
|
||||||
|
- Implementation and tests: `internal/stage/polish.go`,
|
||||||
|
`internal/stage/polish_test.go`,
|
||||||
|
`internal/stage/semantic_contracts_refinement.go`
|
||||||
|
|||||||
@@ -1,74 +1,98 @@
|
|||||||
# Stage: prepare
|
# Stage: prepare
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Materialize all required session inputs into canonical local workspace paths and record input provenance in the session manifest.
|
|
||||||
|
|
||||||
## Inputs and Outputs
|
Materialize canonical current-session inputs before processing stages.
|
||||||
Inputs:
|
|
||||||
- `session.yml` (resolved session config)
|
|
||||||
- `pipeline.resolved.yml` (materialized from resolved pipeline config)
|
|
||||||
- `speakers.yml`
|
|
||||||
- `autocorrect.yml`
|
|
||||||
- `glossary.yml`
|
|
||||||
- audio source:
|
|
||||||
- local (`session.inputs.audio_dir` or `session.inputs.audio_files`), or
|
|
||||||
- S3 (`session.inputs.audio_s3.prefix`)
|
|
||||||
|
|
||||||
Outputs:
|
## Inputs
|
||||||
|
|
||||||
|
- resolved campaign, session, and pipeline configuration
|
||||||
|
- stable input files (`speakers`, `autocorrect`, `glossary`, `players`, `party`)
|
||||||
|
- optional spell-catalog overlay
|
||||||
|
- one resolved local or S3 audio source
|
||||||
|
- enabled configured artifact input requirements for previous-session sources
|
||||||
|
|
||||||
|
## Outputs
|
||||||
|
|
||||||
|
- `inputs/campaign.yml`
|
||||||
- `inputs/session.yml`
|
- `inputs/session.yml`
|
||||||
- `inputs/pipeline.resolved.yml`
|
- `inputs/pipeline.resolved.yml`
|
||||||
- `inputs/speakers.yml`
|
- `inputs/speakers.yml`
|
||||||
- `inputs/autocorrect.yml`
|
- `inputs/autocorrect.yml`
|
||||||
- `inputs/glossary.yml`
|
- `inputs/glossary.yml`
|
||||||
- `audio/*.flac` in session workdir
|
- `inputs/players.yml`
|
||||||
- `manifest.Inputs` records with checksums and source metadata
|
- `inputs/party.yml`
|
||||||
|
- optional `inputs/spell_catalog.json`
|
||||||
|
- `audio/*.flac`
|
||||||
|
- optional `previous/manifest.json`
|
||||||
|
- optional `previous/artifacts/**`
|
||||||
|
- deterministic `manifest.inputs` entries (checksums + provenance)
|
||||||
|
|
||||||
## Boundaries
|
## Key Behavior
|
||||||
Owns:
|
|
||||||
- Input path resolution and validation
|
|
||||||
- Local copy/materialization of configs and audio files
|
|
||||||
- S3 audio download to run-scoped spool, then copy into work audio dir
|
|
||||||
|
|
||||||
Does not own:
|
- validates required config/store state.
|
||||||
- Transcript generation/processing
|
- enforces local audio vs S3 audio mutual exclusivity.
|
||||||
- Archive publish behavior
|
- rejects duplicate explicit local audio sources after resolution.
|
||||||
|
- gives distinct local source paths with the same basename deterministic unique
|
||||||
|
prepared filenames so neither source is overwritten.
|
||||||
|
- materializes S3 audio through spool/cache-aware logic.
|
||||||
|
- materializes a configured spell catalog with checksum and provenance, or
|
||||||
|
safely removes an obsolete canonical spell catalog and its manifest record
|
||||||
|
when the effective input is omitted.
|
||||||
|
- in canonical party mode, copies the validated raw party bytes unchanged and
|
||||||
|
deterministically generates the prepared players projection; legacy mode
|
||||||
|
continues to copy its opaque party and explicit players sources.
|
||||||
|
- scans enabled configured artifact inputs for `narratio.previous_session.artifact.*` requirements.
|
||||||
|
- clears managed `previous/` state on every invocation, then, when requirements exist:
|
||||||
|
- resolves the pointer-selected previous source through the shared resolver;
|
||||||
|
- downloads previous manifest/artifacts;
|
||||||
|
- records previous inputs in `manifest.inputs`.
|
||||||
|
|
||||||
## Config Fields Used
|
Required previous-session inputs fail when unavailable; optional missing inputs
|
||||||
- `session.session_id`
|
are typed skipped results. Committed sources use their exact source-to-destination
|
||||||
- `session.campaign`
|
mapping, while the isolated legacy reader rejects ambiguous fallback matches.
|
||||||
- `session.inputs.speakers_file`
|
|
||||||
- `session.inputs.autocorrect_file`
|
|
||||||
- `session.inputs.glossary_file`
|
|
||||||
- `session.inputs.audio_dir`
|
|
||||||
- `session.inputs.audio_files`
|
|
||||||
- `session.inputs.audio_s3.prefix`
|
|
||||||
- `pipeline.workspace.root`
|
|
||||||
- `pipeline.spool.root`
|
|
||||||
- `pipeline.storage.s3.bucket`
|
|
||||||
- `pipeline.storage.s3.root_prefix`
|
|
||||||
|
|
||||||
## External Adapters Used
|
## Invariants
|
||||||
- Object storage backend (`env.ObjectStore`) for S3 audio list/download when `audio_s3` is configured.
|
|
||||||
|
|
||||||
## State and Manifest Behavior
|
- only `prepare` hydrates canonical `previous/` cache state.
|
||||||
- Ensures workspace layout exists.
|
- managed previous artifacts are stored under `previous/artifacts/**` without
|
||||||
- Writes resolved config and input files to canonical `inputs/` paths.
|
duplicate `artifacts/artifacts/` nesting.
|
||||||
- Records all prepared inputs into `manifest.Inputs` (sorted deterministically by kind/path).
|
- managed `previous/` state represents only the current requirement set.
|
||||||
- For S3 audio, records `S3Bucket`, `S3Key`, `S3Size`, `S3ETag`, and `SpoolPath` in each audio input record.
|
- `manifest.inputs` ordering is deterministic (`kind`, `path`).
|
||||||
|
|
||||||
## Skip and Resume Behavior
|
## Resume Evidence
|
||||||
- Runner-level skip applies when stage already `succeeded` and `--force` is not set.
|
|
||||||
- Stage itself is deterministic/idempotent for unchanged inputs (`copyFileIfChanged`, `writeBytesIfChanged`).
|
|
||||||
|
|
||||||
## Failure Behavior
|
Prepare records a versioned semantic-configuration fingerprint for the
|
||||||
- Fails on missing required files, invalid audio source combinations, no discoverable `.flac` files, duplicate audio basenames, missing object store for S3 mode, or S3 list/download failures.
|
resolved campaign/session selection, local-versus-S3 audio mode and canonical
|
||||||
|
audio names, stable-input ownership/presence, previous-session identity, and
|
||||||
|
the party mode plus canonical players projection version, and the effective
|
||||||
|
previous-artifact requirement set. A change reruns prepare and
|
||||||
|
stales its fixed descendants. Existing successful records without this
|
||||||
|
evidence rerun once when selected.
|
||||||
|
|
||||||
## Tests to Inspect Before Changing
|
Workspace, spool, and cache placement and absolute source relocation are not
|
||||||
- `internal/stage/prepare_test.go`
|
semantic when logical selection, canonical names, and bytes are equivalent.
|
||||||
- `internal/app/session_cli_test.go`
|
The fingerprint deliberately does not read or rehash large audio. Prepared
|
||||||
- `internal/config/load_validate_test.go`
|
input checksums remain the content provenance. Before reusing success, prepare
|
||||||
|
validates every durable prepared copy and compares current stable-input bytes,
|
||||||
|
canonical party and derived-player bytes, local audio membership/checksums, or
|
||||||
|
S3 key/size/entity-tag identity with that provenance. Source relocation with
|
||||||
|
equivalent names and bytes remains reusable; changed or unavailable evidence
|
||||||
|
causes a normal prepare rerun.
|
||||||
|
|
||||||
## Architectural Invariants
|
## Related Contracts And Tests
|
||||||
- `audio_dir`/`audio_files` and `audio_s3` are mutually exclusive.
|
|
||||||
- Audio files must be `.flac`.
|
- [Configuration](../config.md) owns audio selection, stable input fields, and
|
||||||
- Canonical `inputs/*` and `audio/*` paths are the durable source for downstream stages.
|
previous-session settings.
|
||||||
|
- [Operations](../operations.md) owns physical input, audio, spool, cache, and
|
||||||
|
previous-state layout.
|
||||||
|
- [Storage Internals](storage.md) and [Artifact Internals](artifacts.md) explain
|
||||||
|
the internal collaborators.
|
||||||
|
- Implementation and tests: `internal/stage/prepare.go`,
|
||||||
|
`internal/stage/prepare_test.go`,
|
||||||
|
`internal/stage/prepare_resume.go`,
|
||||||
|
`internal/stage/prepare_resume_test.go`,
|
||||||
|
`internal/stage/semantic_contracts_initial.go`,
|
||||||
|
`internal/stage/semantic_contracts_initial_test.go`,
|
||||||
|
`internal/audio/s3_audio_test.go`,
|
||||||
|
`internal/previouscache/*_test.go`
|
||||||
|
|||||||
117
docs/internal/stage-publish.md
Normal file
117
docs/internal/stage-publish.md
Normal file
@@ -0,0 +1,117 @@
|
|||||||
|
# Stage: publish
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
|
||||||
|
Upload run/session outputs to object storage and atomically advance remote current state.
|
||||||
|
|
||||||
|
## Inputs
|
||||||
|
|
||||||
|
- successful preceding stages from the [canonical stage set](overview.md#pipeline-stage-set)
|
||||||
|
- invocation-scoped run files
|
||||||
|
- resolved publish output rules
|
||||||
|
- effective publish locks (static + remote merged lock set), revalidated at the
|
||||||
|
remote commit boundary
|
||||||
|
- durable previous-session cache files when present
|
||||||
|
|
||||||
|
## Outputs
|
||||||
|
|
||||||
|
- uploaded invocation record and selected publish outputs;
|
||||||
|
- uploaded durable previous-session cache files when present;
|
||||||
|
- immutable run-scoped commit manifest; and
|
||||||
|
- current commit pointer, written last.
|
||||||
|
|
||||||
|
Exact remote placement and the operator workflow belong in
|
||||||
|
[Operations](../operations.md#publish-workflow).
|
||||||
|
|
||||||
|
## Key Behavior
|
||||||
|
|
||||||
|
- when publishing or run upload is disabled, completes successfully with no
|
||||||
|
outputs and records explanatory metadata. This is not an explicit self-skip:
|
||||||
|
both manifests record success. Enablement and upload policy are fingerprinted,
|
||||||
|
so changing either automatically makes the prior result non-resumable.
|
||||||
|
- validates prerequisite stage success and object-store availability.
|
||||||
|
- derives a deterministic run-archive allowlist from the validated run
|
||||||
|
`manifest.json`: declared run-local outputs, logs, generated configs, and the
|
||||||
|
manifest itself. Unlisted workspace files are not archive candidates.
|
||||||
|
- opens each archive candidate beneath its archive root without following
|
||||||
|
symlinked ancestors or leaf entries, verifies that it is a regular file and
|
||||||
|
checks a declared checksum when present, then streams the opened descriptor.
|
||||||
|
- derives the durable previous-cache archive from its validated manifest using
|
||||||
|
the same confinement and regular-file checks.
|
||||||
|
- resolves publish output sources through runtime artifact catalog and
|
||||||
|
manifest-aware resolution. Configured Scriptorium outputs are publishable
|
||||||
|
only from validated `current` per-artifact analyze evidence; an incidental
|
||||||
|
canonical file, legacy aggregate output, stale/failed/unselected record, or
|
||||||
|
mismatched path, size, or checksum remains unavailable. This does not change
|
||||||
|
the explicit compatibility policies owned by built-in, extraction, or
|
||||||
|
previous-session sources.
|
||||||
|
- publishes extraction lanes only through explicit configured output rules;
|
||||||
|
neither run-local nor durable Notarius bundles are scanned or uploaded wholesale.
|
||||||
|
- selected artifact filter applies to configured artifact sources only.
|
||||||
|
- locked outputs are skipped intentionally (including required ones).
|
||||||
|
- optional missing outputs are skipped; required missing unlocked outputs fail.
|
||||||
|
- creates one complete immutable source-to-destination mapping before upload;
|
||||||
|
- uploads and verifies every declared immutable object and the commit manifest;
|
||||||
|
- updates `current/commit-pointer.json` exactly once, last; and
|
||||||
|
- does not write the legacy `current/manifest.json` or `current/run_id.txt` pair.
|
||||||
|
- rechecks remote lock state immediately before the pointer update. A newly
|
||||||
|
committed lock aborts selection, leaving any uploaded immutable attempt
|
||||||
|
unselected.
|
||||||
|
- reads the mutable remote lock document through a direct limit-plus-one read
|
||||||
|
capped by `MaxRemoteLockStoreBytes` (1 MiB), retaining the generation returned
|
||||||
|
with the opened body for conditional updates. Oversized lock documents fail
|
||||||
|
before YAML decoding; published artifact payloads do not use this limit.
|
||||||
|
|
||||||
|
## Metadata Signals
|
||||||
|
|
||||||
|
Includes counts/lists for:
|
||||||
|
- run uploads
|
||||||
|
- published output uploads
|
||||||
|
- previous uploads
|
||||||
|
- skipped optional outputs
|
||||||
|
- skipped unselected outputs
|
||||||
|
- locked outputs
|
||||||
|
- remote commit and current-pointer key paths
|
||||||
|
- the run identifier selected by the commit
|
||||||
|
|
||||||
|
## Invariants
|
||||||
|
|
||||||
|
- `current/commit-pointer.json` is the remote commit marker and is written last.
|
||||||
|
- run files, selected outputs, previous-cache files, and the committed session
|
||||||
|
manifest are all declared by an immutable commit under the run prefix.
|
||||||
|
- run and previous uploads contain only manifest-declared regular files opened
|
||||||
|
from verified descriptors; symlinks, special files, replacement races, and
|
||||||
|
undeclared entries are rejected or ignored before uploads begin.
|
||||||
|
- run-local diagnostics, including Notarius receipt and stderr files, are
|
||||||
|
archived only when recorded by the run manifest.
|
||||||
|
- publish locks are not overridden by `--force`; remote locks are revalidated
|
||||||
|
immediately before current-state selection.
|
||||||
|
- post-commit local cleanup is authorized by the committed publish metadata and
|
||||||
|
is durably recorded by the application lifecycle before any local deletion.
|
||||||
|
|
||||||
|
## Resume Semantics
|
||||||
|
|
||||||
|
The versioned semantic fingerprint covers enabled behavior, run-upload policy,
|
||||||
|
normalized source/destination/required output rules, static lock policy, and
|
||||||
|
the remote backend, bucket, region, endpoint, and root-prefix identity. Rule
|
||||||
|
and lock ordering is canonicalized. Credential environment names,
|
||||||
|
path-addressing transport mode, local workspace placement, and run identifiers
|
||||||
|
are excluded. Remote locks remain mutable state and are still revalidated at
|
||||||
|
the commit boundary; semantic evidence does not replace that safety check.
|
||||||
|
|
||||||
|
The commit boundary and cleanup gate are normative architecture invariants; see
|
||||||
|
[Architecture](../policy/architecture.md#publish-commit-boundary).
|
||||||
|
|
||||||
|
## Related Contracts And Tests
|
||||||
|
|
||||||
|
- [Configuration](../config.md#publish-configuration-summary) owns output and
|
||||||
|
static-lock fields.
|
||||||
|
- [Operations](../operations.md#publish-locks) owns remote lock lifecycle and
|
||||||
|
physical remote state.
|
||||||
|
- [Artifact Internals](artifacts.md) explains source resolution and current-state
|
||||||
|
helpers.
|
||||||
|
- Implementation and tests: `internal/stage/publish.go`,
|
||||||
|
`internal/stage/publish_test.go`,
|
||||||
|
`internal/stage/semantic_contracts_delivery.go`,
|
||||||
|
`internal/app/operator_helpers_test.go`, and
|
||||||
|
`internal/app/post_publish_cleanup_test.go`
|
||||||
60
docs/internal/stage-render.md
Normal file
60
docs/internal/stage-render.md
Normal file
@@ -0,0 +1,60 @@
|
|||||||
|
# Stage: render
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
|
||||||
|
Render Markdown transcript artifacts from normalized JSON transcripts via Seriatim.
|
||||||
|
It runs after `trim` and before `extract` in the canonical sequence. Render and
|
||||||
|
extract are independent sibling consumers: replacing render output does not
|
||||||
|
invalidate extraction, but it does invalidate succeeded analysis and delivery
|
||||||
|
records that may consume rendered transcripts.
|
||||||
|
|
||||||
|
## Inputs
|
||||||
|
|
||||||
|
- `narratio.transcript.final` (`transcripts/final.json`)
|
||||||
|
- `narratio.transcript.final_trimmed` (`transcripts/final.trimmed.json`)
|
||||||
|
|
||||||
|
## Outputs
|
||||||
|
|
||||||
|
- `narratio.transcript.final_markdown` -> `transcripts/final.md`
|
||||||
|
- `narratio.transcript.final_trimmed_markdown` -> `transcripts/final.trimmed.md`
|
||||||
|
|
||||||
|
## Key Behavior
|
||||||
|
|
||||||
|
- uses `pipeline.render` settings (enabled/format/title/booleans).
|
||||||
|
- resolves inputs manifest-first, then canonical fallback.
|
||||||
|
- writes run-local outputs first, then materializes canonical session outputs.
|
||||||
|
- records input provenance, output paths, adapter metadata, logs, and generated config refs.
|
||||||
|
- when `pipeline.render.enabled=false`, completes successfully with no outputs
|
||||||
|
and records explanatory metadata. This is not an explicit self-skip: both
|
||||||
|
manifests record success. Because enablement is fingerprinted, enabling
|
||||||
|
render later automatically makes the prior result non-resumable.
|
||||||
|
|
||||||
|
## Failure Semantics
|
||||||
|
|
||||||
|
- missing normalized input fails with normalize rerun guidance.
|
||||||
|
- missing trimmed input fails with trim rerun guidance.
|
||||||
|
- adapter/subprocess failure fails stage.
|
||||||
|
- empty render output files fail validation.
|
||||||
|
|
||||||
|
## Invariants
|
||||||
|
|
||||||
|
- only `format: markdown` is supported.
|
||||||
|
- render stage owns production of built-in Markdown transcript sources.
|
||||||
|
|
||||||
|
## Resume Semantics
|
||||||
|
|
||||||
|
The versioned semantic fingerprint covers enablement, final format, resolved
|
||||||
|
title (including the session-title fallback), timestamp, segment-ID and
|
||||||
|
metadata inclusion, both canonical input identities, and both Markdown output
|
||||||
|
identities. Seriatim's executable, timeout, and report behavior are operational
|
||||||
|
and do not invalidate rendered transcripts. A render-only change leaves the
|
||||||
|
independent `extract` sibling reusable while invalidating their shared
|
||||||
|
downstream consumers.
|
||||||
|
|
||||||
|
## Related Contracts And Tests
|
||||||
|
|
||||||
|
- [Seriatim](../integrations/seriatim.md) owns render subprocess behavior.
|
||||||
|
- [Configuration](../config.md#pipeline) owns render fields and defaults.
|
||||||
|
- Implementation and tests: `internal/stage/render.go`,
|
||||||
|
`internal/stage/render_test.go`,
|
||||||
|
`internal/stage/semantic_contracts_refinement.go`
|
||||||
@@ -1,58 +1,56 @@
|
|||||||
# Stage: transcribe
|
# Stage: transcribe
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Generate per-speaker raw transcripts from prepared audio using WhisperX.
|
|
||||||
|
|
||||||
## Inputs and Outputs
|
Generate raw per-speaker transcripts from prepared audio using WhisperX.
|
||||||
Inputs:
|
|
||||||
- `audio/*.flac` prepared by `prepare`
|
|
||||||
|
|
||||||
Outputs:
|
## Inputs
|
||||||
- `transcripts/raw/<speaker>.json` for each input audio file
|
|
||||||
|
|
||||||
## Boundaries
|
- `audio/*.flac` from `prepare`
|
||||||
Owns:
|
|
||||||
- Discovering prepared audio inputs
|
|
||||||
- Deriving speaker ids from audio basenames
|
|
||||||
- Parallel WhisperX invocation with bounded concurrency
|
|
||||||
- Validating produced JSON and promoting run-local outputs
|
|
||||||
|
|
||||||
Does not own:
|
## Outputs
|
||||||
- Transcript merge/polish/normalize/trim/analyze
|
|
||||||
|
|
||||||
## Config Fields Used
|
- `transcripts/raw/<speaker>.json`
|
||||||
- `session.session_id`
|
|
||||||
- `session.campaign`
|
|
||||||
- `pipeline.workspace.root`
|
|
||||||
- `pipeline.whisperx.transcribe_url`
|
|
||||||
- `pipeline.whisperx.language`
|
|
||||||
- `pipeline.whisperx.timeout`
|
|
||||||
- `pipeline.whisperx.retries`
|
|
||||||
- `pipeline.whisperx.retry_delay`
|
|
||||||
- `pipeline.whisperx.concurrency`
|
|
||||||
|
|
||||||
## External Adapters Used
|
## Key Behavior
|
||||||
- WhisperX adapter (`env.WhisperX.Transcribe`).
|
|
||||||
|
|
||||||
## State and Manifest Behavior
|
- discovers prepared audio from manifest inputs or canonical audio directory.
|
||||||
- Uses run-local output paths under `runs/{run_id}/transcribe/outputs/...` when run layout is enabled.
|
- derives the transcript identity from the prepared `.flac` filename.
|
||||||
- Validates each generated transcript JSON before promotion.
|
- dispatches WhisperX requests through a bounded worker pool.
|
||||||
- Promotes canonical outputs to `transcripts/raw/*.json`.
|
- validates each output as JSON.
|
||||||
- Records per-file metadata (attempts/status/duration/output path) in stage metadata.
|
- writes run-local outputs then materializes canonical transcript outputs only
|
||||||
|
after every planned request succeeds.
|
||||||
|
|
||||||
## Skip and Resume Behavior
|
## Invariants
|
||||||
- Runner-level skip applies for previously succeeded stage unless forced.
|
|
||||||
- On forced upstream reruns, downstream succeeded stages can be marked `stale` by runner logic.
|
|
||||||
|
|
||||||
## Failure Behavior
|
- prepared audio identities must be unique; prepare disambiguates distinct
|
||||||
- Fails if no prepared audio exists, duplicate speaker basenames are detected, adapter output path mismatches expected path, any output JSON is invalid, or one worker fails.
|
source paths that share a basename.
|
||||||
- Cancels in-flight workers after first terminal error.
|
- output path returned by adapter must match requested output path.
|
||||||
|
- an empty adapter result path means the requested path; adapters cannot select
|
||||||
|
an alternate destination.
|
||||||
|
- each successful output is validated before stage success, and cancellation or
|
||||||
|
incomplete dispatch cannot be reported as a successful result.
|
||||||
|
|
||||||
## Tests to Inspect Before Changing
|
## Resume Evidence
|
||||||
- `internal/stage/transcribe_test.go`
|
|
||||||
- `internal/app/whisperx_wiring_test.go`
|
|
||||||
|
|
||||||
## Architectural Invariants
|
Transcribe records a versioned semantic-configuration fingerprint containing
|
||||||
- Speaker identity is derived from `.flac` basename and must be unique.
|
the Narratio-visible WhisperX service URL and recognition language. Changes to
|
||||||
- Every successful speaker output must be valid JSON before promotion.
|
either rerun transcription and stale its fixed descendants while leaving
|
||||||
- Canonical raw transcript set is the only supported merge input surface.
|
prepare reusable. Retry count/delay, concurrency, timeout, credentials, and
|
||||||
|
diagnostic locations are operational and do not change this evidence.
|
||||||
|
|
||||||
|
WhisperX models or private service configuration not exposed by Narratio's
|
||||||
|
adapter contract cannot be fingerprinted; use `--force` after changing them.
|
||||||
|
An existing successful transcribe record without evidence reruns once when
|
||||||
|
selected.
|
||||||
|
|
||||||
|
## Related Contracts And Tests
|
||||||
|
|
||||||
|
- [WhisperX](../integrations/whisperx.md) owns HTTP, retry, timeout, and
|
||||||
|
cancellation semantics.
|
||||||
|
- [Configuration](../config.md#pipeline) owns concurrency and other
|
||||||
|
operator-selected values.
|
||||||
|
- Implementation and tests: `internal/stage/transcribe.go`,
|
||||||
|
`internal/stage/transcribe_test.go`,
|
||||||
|
`internal/stage/semantic_contracts_initial.go`, and
|
||||||
|
`internal/stage/semantic_contracts_initial_test.go`
|
||||||
|
|||||||
@@ -1,75 +1,57 @@
|
|||||||
# Stage: trim
|
# Stage: trim
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Optionally trim the normalized transcript to session bounds; always produce a durable trimmed transcript.
|
|
||||||
|
|
||||||
## Inputs and Outputs
|
Produce a final-trimmed transcript. By default, the stage generates bounds and
|
||||||
Inputs:
|
applies a bounds-driven trim.
|
||||||
- `transcripts/normalized.json`
|
|
||||||
|
|
||||||
Outputs:
|
## Inputs
|
||||||
- `transcripts/trimmed.json` (or configured trim output path)
|
|
||||||
|
- `transcripts/final.json`
|
||||||
|
|
||||||
|
## Outputs
|
||||||
|
|
||||||
|
- `transcripts/final.trimmed.json` (or configured trim output path)
|
||||||
- when trim enabled: `artifacts/session_bounds.json`
|
- when trim enabled: `artifacts/session_bounds.json`
|
||||||
|
|
||||||
## Boundaries
|
## Key Behavior
|
||||||
Owns:
|
|
||||||
- Trim-enabled switch behavior
|
|
||||||
- Bounds generation via Scriptorium artifact run
|
|
||||||
- Bounds validation against normalized transcript
|
|
||||||
- Keep-selector derivation and Seriatim trim invocation
|
|
||||||
- Copy-through behavior when disabled or bounds indicate unchanged transcript
|
|
||||||
|
|
||||||
Does not own:
|
When `trim.enabled=true`:
|
||||||
- Upstream normalization
|
- runs Scriptorium bounds artifact generation;
|
||||||
- Downstream artifact analysis
|
- optionally runs render-debug output generation;
|
||||||
|
- validates bounds payload against transcript;
|
||||||
|
- derives keep selector;
|
||||||
|
- either copies unchanged transcript or runs Seriatim trim;
|
||||||
|
- validates trimmed transcript and materializes bounds output.
|
||||||
|
|
||||||
## Config Fields Used
|
When `trim.enabled=false`:
|
||||||
- `session.session_id`
|
- copies normalized transcript to trimmed output.
|
||||||
- `session.campaign`
|
|
||||||
- `pipeline.workspace.root`
|
|
||||||
- `pipeline.trim.enabled`
|
|
||||||
- `pipeline.trim.output_path`
|
|
||||||
- `pipeline.trim.bounds.prompt_id`
|
|
||||||
- `pipeline.trim.bounds.profile_id`
|
|
||||||
- `pipeline.trim.bounds.timeout`
|
|
||||||
- `pipeline.trim.bounds.output_path`
|
|
||||||
- `pipeline.trim.bounds.transcript_input_name`
|
|
||||||
- `pipeline.trim.bounds.render_debug`
|
|
||||||
- `pipeline.trim.bounds.render_output_path`
|
|
||||||
- `pipeline.seriatim.binary`
|
|
||||||
- `pipeline.seriatim.timeout`
|
|
||||||
- `pipeline.scriptorium.binary`
|
|
||||||
- `pipeline.scriptorium.config_path`
|
|
||||||
- `pipeline.scriptorium.timeout`
|
|
||||||
|
|
||||||
## External Adapters Used
|
## Invariants
|
||||||
- Scriptorium adapter:
|
|
||||||
- optional `RenderArtifact` for bounds debug render
|
|
||||||
- `RunArtifact` for bounds output
|
|
||||||
- Seriatim adapter:
|
|
||||||
- `Trim` when bounds indicate trimming is required
|
|
||||||
|
|
||||||
## State and Manifest Behavior
|
- normalized transcript is required input.
|
||||||
- Reads normalized transcript from normalize manifest outputs when available; falls back to canonical path.
|
- bounds output exists only in enabled trim path.
|
||||||
- Uses run-local outputs/logs/reports/config/scratch paths when run layout is enabled.
|
- render-debug output is diagnostic and not a declared stage output.
|
||||||
- Promotes canonical trimmed transcript; promotes session bounds when trim enabled.
|
|
||||||
- Records bounds diagnostics, trim action, keep selector, and adapter metadata.
|
|
||||||
|
|
||||||
## Skip and Resume Behavior
|
## Resume Semantics
|
||||||
- Runner-level skip applies when already succeeded and not forced.
|
|
||||||
- Forced reruns can stale downstream succeeded stages.
|
|
||||||
- When `trim.enabled=false`, stage still succeeds by copying normalized to trimmed output.
|
|
||||||
|
|
||||||
## Failure Behavior
|
The versioned semantic fingerprint covers enablement, the bounds prompt and
|
||||||
- Fails on missing/invalid normalized transcript.
|
profile identifiers, the Scriptorium configuration identity, transcript input
|
||||||
- With trim enabled, fails on missing adapters/config, bounds generation/validation errors, invalid bounds JSON, invalid range/segment ids, trim adapter failures, or invalid trimmed output.
|
name, sticky session variable, bounds and trimmed output identities, and the
|
||||||
|
Seriatim trim operation. Diagnostic bounds rendering, diagnostic output paths,
|
||||||
|
timeouts, executable paths, and optional reports are operational and do not
|
||||||
|
invalidate the canonical trimmed transcript.
|
||||||
|
|
||||||
## Tests to Inspect Before Changing
|
Narratio cannot inspect prompt, profile, or configuration content that
|
||||||
- `internal/stage/trim_test.go`
|
Scriptorium or Seriatim privately resolves behind a stable identifier. Force
|
||||||
- `internal/adapters/scriptorium/subprocess_test.go`
|
`trim` after changing such private content without changing its identifier.
|
||||||
- `internal/adapters/seriatim/subprocess_test.go`
|
|
||||||
|
|
||||||
## Architectural Invariants
|
## Related Contracts And Tests
|
||||||
- Trim never falls back to processed transcript; normalized transcript is required input.
|
|
||||||
- `session_bounds` output exists only for enabled trim path.
|
- [Scriptorium](../integrations/scriptorium.md) owns bounds generation and
|
||||||
- Render-debug artifacts are diagnostics and not declared stage outputs.
|
debug-render subprocess behavior.
|
||||||
|
- [Seriatim](../integrations/seriatim.md) owns transcript trimming behavior.
|
||||||
|
- [Configuration](../config.md#pipeline) owns trim fields and defaults.
|
||||||
|
- Implementation and tests: `internal/stage/trim.go`,
|
||||||
|
`internal/stage/trim_test.go`,
|
||||||
|
`internal/stage/semantic_contracts_refinement.go`
|
||||||
|
|||||||
@@ -1,71 +1,64 @@
|
|||||||
# Internal: Storage
|
# Internal: Storage
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Document Narratio's remote storage backend contracts and implementations under `internal/adapters/storage`.
|
|
||||||
|
|
||||||
## Inputs and outputs
|
Explain the object-store interface and S3 implementation used by Narratio.
|
||||||
Inputs:
|
Remote key layout and lifecycle belong in [Operations](../operations.md), while
|
||||||
- Resolved storage config (`pipeline.storage.*`).
|
operator-selected storage fields and credential mechanisms belong in
|
||||||
- Bucket-relative object keys and local file paths from stage/app orchestration.
|
[Configuration](../config.md).
|
||||||
|
|
||||||
Outputs:
|
## Primary Contract
|
||||||
- Listed/downloaded/uploaded object metadata (`ObjectInfo`).
|
|
||||||
- Existence checks and storage-layer errors.
|
|
||||||
|
|
||||||
## Boundaries
|
`storage.ObjectStore` interface:
|
||||||
Owns:
|
|
||||||
- Remote object-store interface and implementation details.
|
|
||||||
- S3 client wiring and API calls.
|
|
||||||
- Object key normalization and upload/download/list primitives.
|
|
||||||
|
|
||||||
Does not own:
|
- `List(ctx, prefix)`
|
||||||
- Session/run prefix semantics.
|
- `Read(ctx, key)` returns an object body and the generation observed with it
|
||||||
- Archive commit order semantics.
|
- `Download(ctx, key, localPath)`
|
||||||
- Manifest updates.
|
- `Upload(ctx, localPath, key, opts)`
|
||||||
|
- `UploadConditional(ctx, source, key, opts, condition)`
|
||||||
|
- `Exists(ctx, key)`
|
||||||
|
|
||||||
## Config fields used
|
Key invariant:
|
||||||
- `pipeline.storage.backend`
|
- callers pass full bucket-relative keys;
|
||||||
- `pipeline.storage.s3.bucket`
|
- storage implementations do not infer campaign/session/run prefixes.
|
||||||
- `pipeline.storage.s3.region`
|
|
||||||
- `pipeline.storage.s3.endpoint`
|
|
||||||
- `pipeline.storage.s3.force_path_style`
|
|
||||||
- `pipeline.storage.s3.access_key_id_env`
|
|
||||||
- `pipeline.storage.s3.secret_access_key_env`
|
|
||||||
|
|
||||||
## External adapters used
|
`ReadObjectBounded` is the shared mechanism for small control objects. It opens
|
||||||
Storage package contracts:
|
one object version, returns the metadata observed with that body, rejects an
|
||||||
- `ObjectStore` (active remote object-store boundary): `List`, `Download`, `Upload`, `Exists`.
|
oversized known size before transfer, and still performs a context-aware
|
||||||
- `Backend` (archive request boundary): currently implemented with `NoopBackend` only.
|
limit-plus-one read. It closes the body on every exit. Callers own the policy
|
||||||
|
limit and add the control-object category to errors; this helper is not used for
|
||||||
|
large artifact payloads.
|
||||||
|
|
||||||
Implementations:
|
## Composition
|
||||||
- `S3Backend`: AWS SDK-backed `ObjectStore` implementation.
|
|
||||||
- `FakeBackend`: deterministic test `ObjectStore` and archive backend.
|
|
||||||
- `NoopBackend`: deterministic no-op archive backend for compatibility wiring.
|
|
||||||
|
|
||||||
## State and manifest behavior
|
`NewObjectStoreFromConfig` constructs the S3-backed implementation from
|
||||||
- Storage implementations are stateless with respect to manifest/session lifecycle.
|
resolved configuration. The application loads configured filesystem secrets
|
||||||
- Caller supplies fully-qualified bucket-relative keys.
|
before calling it. The storage adapter consumes already-resolved values; it does
|
||||||
- Storage layer does not infer campaign/session/run/root-prefix semantics.
|
not own discovery, defaults, or configuration validation.
|
||||||
- Caller controls publish ordering; storage layer executes individual operations in the order invoked.
|
|
||||||
|
|
||||||
## Skip and resume behavior
|
## S3 Backend Behavior
|
||||||
- No storage-level skip/resume behavior.
|
|
||||||
- Skip/resume decisions are made by stage/app logic before storage calls occur.
|
|
||||||
|
|
||||||
## Failure behavior
|
- normalizes object keys.
|
||||||
- `NewObjectStoreFromConfig` fails when no remote backend is configured or required S3 config is missing.
|
- `List` paginates and returns normalized `ObjectInfo`.
|
||||||
- `S3Backend` constructor fails when required bucket is missing or AWS client setup fails.
|
- A truncated S3 listing must supply a new, non-empty continuation token;
|
||||||
- CRUD operations return contextual errors (including not-found behavior via `Exists`).
|
otherwise listing fails with bucket and prefix context instead of looping.
|
||||||
- Key normalization is applied before operations (`\\` to `/`, leading slash trimmed).
|
- `Download` writes local files with parent directory creation.
|
||||||
|
- `Upload` streams local file and returns remote metadata.
|
||||||
|
- `Read` binds a returned body to its S3 ETag. `UploadConditional` maps an ETag
|
||||||
|
match or absence precondition directly to the provider request and reports a
|
||||||
|
failed precondition without performing a local check-then-write replacement.
|
||||||
|
- `Exists` maps not-found responses to `false`.
|
||||||
|
|
||||||
## Tests to inspect before changing
|
## Invariants
|
||||||
- `internal/adapters/storage/factory_test.go`
|
|
||||||
- `internal/adapters/storage/s3_backend_test.go`
|
|
||||||
- `internal/adapters/storage/fake_test.go`
|
|
||||||
- `internal/adapters/storage/keys_test.go`
|
|
||||||
- `internal/adapters/storage/archive.go` + consumers in stage tests (`prepare`, `archive`)
|
|
||||||
|
|
||||||
## Architectural invariants
|
- storage layer is stateless regarding manifest/stage progression.
|
||||||
- Callers pass full bucket-relative keys.
|
- bounded reads never retain more than the caller's limit plus one byte and do
|
||||||
- Storage backends must not prepend or infer narratio prefixes.
|
not replace owner-specific size policy.
|
||||||
- Remote transport details remain isolated to storage adapter implementations.
|
- publish ordering semantics are owned by stage/app code, not storage adapters.
|
||||||
|
|
||||||
|
## Implementation And Tests
|
||||||
|
|
||||||
|
- Contract and S3 adapter: `internal/adapters/storage`
|
||||||
|
- Composition: `internal/app/object_store.go`
|
||||||
|
- Tests: `internal/adapters/storage/*_test.go`,
|
||||||
|
`internal/app/object_store_test.go`
|
||||||
|
|||||||
@@ -1,68 +1,94 @@
|
|||||||
# Workspace internals
|
# Internal: Workspace
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Define the local durable and run-local workspace model used by stages, manifests, resume, and archive.
|
|
||||||
|
|
||||||
## Inputs and Outputs
|
Explain the helpers that construct local session and run paths, coordinate
|
||||||
Inputs:
|
single-writer access, and confine cleanup. The authoritative physical layout and
|
||||||
- `pipeline.workspace.root`
|
retention workflow belong in [Operations](../operations.md#local-state-layout).
|
||||||
- `session.campaign`
|
|
||||||
- `session.session_id`
|
|
||||||
- generated `run_id`
|
|
||||||
|
|
||||||
Outputs:
|
## Path Ownership
|
||||||
- Session manifest at `{workspace.root}/work/{campaign}/{session_id}/manifest.json`
|
|
||||||
- Run manifest at `{workspace.root}/work/{campaign}/{session_id}/runs/{run_id}/manifest.json`
|
|
||||||
- Canonical durable session directories and run-local stage trees
|
|
||||||
|
|
||||||
## Boundaries
|
`internal/artifacts` owns canonical session, run, spool, cache, and
|
||||||
Owns:
|
previous-cache path construction. `SessionPathsFor` provides the session-scoped
|
||||||
- Session-level path layout (`inputs/`, `audio/`, `transcripts/`, `artifacts/`, `reports/`, `logs/`, `config/`, `current/`, `runs/`)
|
path model, and layout creation goes through `EnsureLayoutFor`. Callers should
|
||||||
- Run-local stage sandbox layout under `runs/{run_id}/{stage}/`
|
consume those helpers instead of rebuilding relative paths.
|
||||||
- Session lock acquisition/release (`.lock`)
|
|
||||||
|
|
||||||
Does not own:
|
`internal/pathsafe` validates relative destinations. `internal/fileops` opens
|
||||||
- Stage business logic
|
cleanup roots and their descendants through no-follow directory handles before
|
||||||
- Remote archive semantics (documented in `stage-archive.md`)
|
removing them.
|
||||||
- CLI argument parsing
|
|
||||||
|
|
||||||
## Config Fields Used
|
`internal/fileops` owns the ordinary workspace mode contract. On POSIX,
|
||||||
- `pipeline.workspace.root`
|
`WorkspaceDirectoryMode` is setgid `02775` and `WorkspaceFileMode` is `0664`.
|
||||||
- `pipeline.workspace.cleanup_after_archive`
|
`EnsureWorkspaceDirectory` reapplies the directory mode after creation so a
|
||||||
- `pipeline.spool.root`
|
restrictive umask cannot remove group access, while retaining existing ownership
|
||||||
- `pipeline.spool.delete_audio_after_archive`
|
and group. Credential paths are outside this contract; the platform-specific
|
||||||
- `session.campaign`
|
operational requirements are in [Operations](../operations.md#workspace-permissions).
|
||||||
- `session.session_id`
|
|
||||||
|
|
||||||
## External Adapters Used
|
## Run-Local Stage Layout
|
||||||
None directly in this subsystem. Stages may use object storage adapters and then write local outputs into this layout.
|
|
||||||
|
|
||||||
## State and Manifest Behavior
|
`internal/stage/run_local.go` maps stage outputs and diagnostics into an
|
||||||
- Session state is persisted in the session manifest (`manifest.Manifest`).
|
invocation-scoped layout. Successful outputs are validated and atomically
|
||||||
- Invocation history is persisted per run in run manifests under `runs/{run_id}/manifest.json`.
|
materialized into canonical session paths before stage success. Managed
|
||||||
- During each run, stage outputs are often written run-local first (`runs/{run_id}/{stage}/outputs/...`) and promoted to canonical session paths after stage success.
|
previous-session cache paths remain session-durable and are never redirected
|
||||||
- `manifest.Artifacts` entries record `ProducerRunID` for durable outputs.
|
into run-local output space.
|
||||||
- For S3 audio sessions, `prepare` records spool/work paths and S3 provenance in `manifest.Inputs`.
|
|
||||||
|
|
||||||
## Skip and Resume Behavior
|
Extraction uses run-local receipt, stderr, and output-root helpers, then
|
||||||
- Skip/resume decisions are made in `internal/app` (`run_control.go`, `resume.go`) using stage status in the session manifest.
|
promotes the validated external bundle to the unique immutable Notarius bundle
|
||||||
- `--force` reruns selected stages and marks downstream previously-succeeded stages as `stale`.
|
path supplied by `internal/artifacts`. `internal/fileops.PromoteDirectory`
|
||||||
- Workspace layout is idempotent (`EnsureLayoutFor`) and reused across runs.
|
copies only regular files and directories to a same-filesystem temporary
|
||||||
|
sibling. Source traversal uses confined directory handles and identity checks
|
||||||
|
so replacing an inspected root, directory, or file is rejected rather than
|
||||||
|
followed. The completed tree is atomically renamed without replacing an
|
||||||
|
existing destination. Exact physical paths belong in
|
||||||
|
[Operations](../operations.md#extraction-workflow).
|
||||||
|
|
||||||
## Failure Behavior
|
## Locking
|
||||||
- Failures preserve manifests and run-local files for inspection.
|
|
||||||
- Lock conflicts fail fast via `ErrLockConflict`.
|
|
||||||
- Cleanup can fail post-archive; failure is recorded in archive stage metadata and returned by the run.
|
|
||||||
|
|
||||||
## Tests to Inspect Before Changing
|
`artifacts.LocalStore` enforces the single-writer session lock via an
|
||||||
- `internal/artifacts/local_test.go`
|
operating-system lock held on `.lock` (`ErrLockConflict` on contention). The
|
||||||
- `internal/stage/run_local_test.go`
|
file retains owner metadata after release or process death; its existence is
|
||||||
- `internal/app/run_control_test.go`
|
not evidence that a lock is active. Command and restore flows wait for this
|
||||||
- `internal/app/resume_run_stage_test.go`
|
lock only while their context remains active, and report a release failure.
|
||||||
- `internal/app/post_archive_cleanup_test.go`
|
|
||||||
|
|
||||||
## Architectural Invariants
|
## Cleanup Semantics
|
||||||
- Session root is campaign-aware: `{workspace.root}/work/{campaign}/{session_id}`.
|
|
||||||
- Run roots are always nested: `runs/{run_id}` under the session root.
|
Automatic post-publish cleanup:
|
||||||
- Run-local output promotion must end in canonical session paths.
|
|
||||||
- Cleanup only targets run-scoped directories and must never delete configured root directories.
|
- is created only after a successful publish commit with complete publish
|
||||||
|
metadata, then is persisted before any deletion;
|
||||||
|
- requires `uploaded=true`, a remote commit key, and a current commit-pointer
|
||||||
|
key in publish metadata;
|
||||||
|
- consumes the resolved cleanup policy described in
|
||||||
|
[Configuration](../config.md);
|
||||||
|
- refuses unsafe deletes (root delete, out-of-root delete, and symlinked
|
||||||
|
ancestors or entries);
|
||||||
|
- retries any recorded incomplete target on later invocations even when no
|
||||||
|
publish work is selected. Missing targets are a successful, idempotent
|
||||||
|
cleanup result only after the completion evidence is saved.
|
||||||
|
|
||||||
|
Manual cleanup uses the same root-confined deletion mechanism. Invocation
|
||||||
|
syntax and exact deletion scope belong in [CLI](../cli.md#clean) and
|
||||||
|
[Operations](../operations.md#cleanup).
|
||||||
|
|
||||||
|
## Invariants
|
||||||
|
|
||||||
|
- campaign-aware session root is mandatory.
|
||||||
|
- manifest-driven stage state is durable across runs.
|
||||||
|
- cleanup guardrails prevent destructive root/out-of-scope deletion.
|
||||||
|
- ordinary workspace paths retain group-writable directory and file modes across
|
||||||
|
nested creation, replacement, and Notarius promotion.
|
||||||
|
|
||||||
|
## Implementation And Tests
|
||||||
|
|
||||||
|
- Path model and local store: `internal/artifacts/paths.go`,
|
||||||
|
`internal/artifacts/local.go`
|
||||||
|
- Run-local materialization: `internal/stage/run_local.go`
|
||||||
|
- Immutable bundle promotion: `internal/fileops/directory.go`
|
||||||
|
- Workspace modes: `internal/fileops/modes.go`
|
||||||
|
- Cleanup confinement: `internal/fileops/cleanup.go`,
|
||||||
|
`internal/app/cleanup_targets.go`, `internal/app/post_publish_cleanup.go`
|
||||||
|
- Tests: `internal/artifacts/paths_model_test.go`,
|
||||||
|
`internal/artifacts/local_test.go`, `internal/stage/run_local_test.go`,
|
||||||
|
`internal/fileops/directory_test.go`, `internal/fileops/modes_posix_test.go`,
|
||||||
|
`internal/fileops/cleanup_test.go`, `internal/app/cleanup_targets_test.go`,
|
||||||
|
`internal/app/post_publish_cleanup_test.go`
|
||||||
|
|||||||
@@ -1,149 +1,574 @@
|
|||||||
# Operations
|
# Operations Guide
|
||||||
|
|
||||||
This guide describes the implemented operator lifecycle for Narratio.
|
Operator workflow for running, recovering, and publishing Narratio sessions.
|
||||||
|
|
||||||
For field-level configuration, see [docs/config.md](./config.md). For full command/flag reference, see [docs/cli.md](./cli.md).
|
For command syntax, see [docs/cli.md](./cli.md). For field-level config, see [docs/config.md](./config.md).
|
||||||
|
|
||||||
## Normal workflow (S3-first path)
|
## Campaign and Session Selection
|
||||||
|
|
||||||
1. Upload session `.flac` files to object storage under the session audio prefix.
|
Campaign selection priority:
|
||||||
2. Run Narratio:
|
|
||||||
|
- `--campaign-file`
|
||||||
|
- `--campaign`
|
||||||
|
- `pipeline.campaigns.default_campaign_id`
|
||||||
|
|
||||||
|
Session source priority:
|
||||||
|
|
||||||
|
- `--session`
|
||||||
|
- local default search paths
|
||||||
|
- remote session object (S3) when local session file is not found and storage is configured
|
||||||
|
|
||||||
|
## Session Initialization
|
||||||
|
|
||||||
|
Use `session init` to generate a concrete session file for local or remote use.
|
||||||
|
|
||||||
|
Local file:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio run --session-id 2026-04-04
|
narratio session init 2026-04-04 --output ./session.yml --date 2026-04-04 --title "Session 12"
|
||||||
```
|
```
|
||||||
|
|
||||||
3. Read success output:
|
Remote session object:
|
||||||
- `narratio run: session <session_id>; executed=<n> skipped=<n>; manifest=<path>`
|
|
||||||
- use `manifest=<path>` with `status` for inspection.
|
|
||||||
|
|
||||||
Notes:
|
```bash
|
||||||
- default config/session discovery applies unless `--config` and `--session` are passed.
|
narratio session init 2026-04-04 --remote --force
|
||||||
- S3 audio mode requires `session.inputs.audio_s3.prefix` and valid object-store access.
|
```
|
||||||
|
|
||||||
## Local filesystem layout and state artifacts
|
If `campaign.yml` sets `session_template_file`, `session init` renders it. Template variables must resolve to concrete values.
|
||||||
|
|
||||||
|
Campaigns must provide stable input files for speakers, autocorrect, glossary,
|
||||||
|
players, and party, and may provide an optional spell-catalog overlay. Session
|
||||||
|
files may override those paths for one session. The `prepare` stage
|
||||||
|
materializes them under `inputs/`; configured consumers use the prepared files,
|
||||||
|
never the original campaign or session source paths. Field definitions and
|
||||||
|
source IDs are in [Configuration](./config.md#notarius-reference-bindings).
|
||||||
|
|
||||||
|
## Standard Session Workflow
|
||||||
|
|
||||||
|
1. Select pipeline/campaign/session config.
|
||||||
|
2. Validate session readiness:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session validate 2026-04-04
|
||||||
|
```
|
||||||
|
|
||||||
|
3. (Optional) inspect stage decisions:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session plan 2026-04-04
|
||||||
|
```
|
||||||
|
|
||||||
|
4. Run the pipeline:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run 2026-04-04
|
||||||
|
```
|
||||||
|
|
||||||
|
5. Check state:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session status 2026-04-04
|
||||||
|
```
|
||||||
|
|
||||||
|
Run, plan, and status output identify the resolved pipeline profile (or `none`)
|
||||||
|
and effective configuration digest. Status distinguishes the current resolved
|
||||||
|
value from the last value persisted in the session manifest, which helps
|
||||||
|
diagnose profile switches without changing resume authority.
|
||||||
|
|
||||||
|
Before switching an operational profile, compare its effective meaning with the
|
||||||
|
current selection through `narratio config diff <left-profile> <right-profile>`.
|
||||||
|
The command is read-only and succeeds whether it finds differences or not. Its
|
||||||
|
sorted records describe defaulted, expanded concrete configuration—not source
|
||||||
|
file layout—so it can be used to review model, artifact, and publish changes
|
||||||
|
without creating a session or run. Select the same campaign explicitly when
|
||||||
|
profiles could resolve different campaign paths; see the [CLI reference](cli.md#config-validate-config-show-config-sources-and-config-diff) for syntax and record format.
|
||||||
|
|
||||||
|
## Stage Execution and Continuation Behavior
|
||||||
|
|
||||||
|
Canonical stage order:
|
||||||
|
|
||||||
|
1. `prepare`
|
||||||
|
2. `transcribe`
|
||||||
|
3. `merge`
|
||||||
|
4. `polish`
|
||||||
|
5. `normalize`
|
||||||
|
6. `trim`
|
||||||
|
7. `render`
|
||||||
|
8. `extract`
|
||||||
|
9. `analyze`
|
||||||
|
10. `publish`
|
||||||
|
11. `notify`
|
||||||
|
|
||||||
|
Execution rules:
|
||||||
|
|
||||||
|
- succeeded stages are skipped unless `--force` is set; stages with semantic
|
||||||
|
configuration contracts additionally require matching versioned evidence,
|
||||||
|
and missing legacy evidence causes a safe one-time rerun;
|
||||||
|
- `run` continues interrupted or partially completed sessions by running non-succeeded stages;
|
||||||
|
- forcing a stage marks succeeded transitive dependents as `stale` before the
|
||||||
|
replacement runs; render and extract are independent siblings; and
|
||||||
|
- an executed failure, changed self-skip, or success that replaces a different
|
||||||
|
effective outcome uses the same fixed dependency relation. A
|
||||||
|
repeated self-skip with the same reason and no outputs is stable and does not
|
||||||
|
perpetually rerun dependent work.
|
||||||
|
|
||||||
|
Every aggregate stage except analyze currently provides semantic-configuration
|
||||||
|
evidence; analyze retains its more precise per-artifact fingerprints and
|
||||||
|
validator.
|
||||||
|
Prepare additionally validates current stable/local/S3 source identity and the
|
||||||
|
checksums of its durable prepared copies before reuse. Changed bytes, audio
|
||||||
|
membership, S3 object identity, or missing/tampered copies rerun prepare and
|
||||||
|
its fixed descendants without requiring `--force`. Changing prepare selection
|
||||||
|
semantics likewise reruns all fixed descendants; changing
|
||||||
|
WhisperX language/service identity reuses prepare; changing a Seriatim merge
|
||||||
|
transformation reuses prepare and transcribe; and changing an Audita model
|
||||||
|
reuses prepare, transcribe, and merge while rebuilding transcript refinement.
|
||||||
|
A trim change invalidates both render and extract through the fixed dependency
|
||||||
|
relation, while a render-only change preserves the extract sibling.
|
||||||
|
|
||||||
|
Operational timeouts, retry/concurrency tuning, executable paths,
|
||||||
|
workspace/cache/spool placement, reports, diagnostics, and secret values are
|
||||||
|
excluded. Configuration, models, prompts, modules, or resources loaded
|
||||||
|
privately inside external tools remain unobservable to Narratio. If their
|
||||||
|
contents change behind the same configured identifier, explicitly force the
|
||||||
|
affected stage.
|
||||||
|
|
||||||
|
An explicit self-skip is a durable `skipped` stage outcome that later runs
|
||||||
|
reconsider. It differs from successful no-output execution: disabled `render`
|
||||||
|
and `publish`, and absent or no-executable `analyze`, record `succeeded` with
|
||||||
|
metadata and no outputs. Ordinary later runs reuse those successful results;
|
||||||
|
force the affected stage after enabling or configuring it. Optional artifact
|
||||||
|
inputs are omitted only from the consuming artifact invocation and do not make
|
||||||
|
the stage self-skip.
|
||||||
|
|
||||||
|
Single-stage execution:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run-stage normalize 2026-04-04 --force
|
||||||
|
```
|
||||||
|
|
||||||
|
Contiguous bounded execution uses inclusive canonical endpoints:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session plan 2026-04-04 --from extract --through analyze --force
|
||||||
|
narratio run 2026-04-04 --from extract --through analyze --force
|
||||||
|
```
|
||||||
|
|
||||||
|
Omitting `--from` selects from `prepare`; omitting `--through` selects through
|
||||||
|
`notify`. Force applies only within the selected range. Repeating `--from`,
|
||||||
|
`--through`, or `--force` is rejected instead of resolving by argument order.
|
||||||
|
The plan command uses the same selection contract and prints only the selected
|
||||||
|
range. Planning is read-only: it clones the loaded manifest, models selected
|
||||||
|
stage transitions and invalidation in memory, and invokes resume validation
|
||||||
|
without writing the manifest, creating run directories, materializing files,
|
||||||
|
or invoking pipeline adapters. Analyze detail separates explicit targets,
|
||||||
|
prerequisite rebuilds, scheduled execution, and current reuse. This lets a
|
||||||
|
coarsely stale aggregate analyze stage show zero artifact executions when its
|
||||||
|
selected artifact evidence is still semantically current.
|
||||||
|
|
||||||
|
Before a bounded run or plan whose range starts after `prepare`, every excluded
|
||||||
|
prefix stage must already have a session-manifest status of `succeeded` or
|
||||||
|
`skipped`. Narratio reports the first absent, pending, running, failed, stale,
|
||||||
|
or interrupted prerequisite without creating a run record or changing session
|
||||||
|
state. Widen `--from` to include that stage, or recover it explicitly before
|
||||||
|
retrying. Excluded prefix stages are not resume-validated or repaired as part
|
||||||
|
of the bounded invocation; selected stages still reject missing, unsafe, or
|
||||||
|
manifest-inconsistent inputs at their owning boundary.
|
||||||
|
|
||||||
|
Stages after `--through` are not prerequisites and are never scheduled by the
|
||||||
|
bounded invocation. A selected forced stage can mark one of those succeeded
|
||||||
|
dependents stale through the fixed invalidation relation, but the dependent
|
||||||
|
does not execute until a later invocation selects it. Production composition
|
||||||
|
likewise initializes only collaborators needed by the selected range and
|
||||||
|
shared session lifecycle. In particular, render does not require Notarius or
|
||||||
|
Scriptorium, extract does not require Scriptorium, and analyze does not require
|
||||||
|
the transcription, Seriatim, Audita, or Notarius adapters.
|
||||||
|
|
||||||
|
For the common post-transcript development loop, use:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio regenerate-artifacts 2026-04-04
|
||||||
|
narratio regenerate-artifacts 2026-04-04 --artifacts session_recap,player_handout
|
||||||
|
```
|
||||||
|
|
||||||
|
This command is a transparent expansion to a forced bounded `run` from
|
||||||
|
`extract` through `analyze`. Extraction always rebuilds its complete configured
|
||||||
|
bundle. Analysis rebuilds the selected targets and their required analysis
|
||||||
|
prerequisites, or uses the normal default selection when no artifact names are
|
||||||
|
given. The command does not run publish or notify; delivery remains a separate
|
||||||
|
operator action.
|
||||||
|
|
||||||
|
Inspect current artifact evidence, then publish explicitly when the regenerated
|
||||||
|
set is ready:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session artifacts 2026-04-04
|
||||||
|
narratio publish 2026-04-04
|
||||||
|
```
|
||||||
|
|
||||||
|
If planning or execution reports stale, missing, failed, legacy, or tampered
|
||||||
|
analysis evidence, regenerate the affected target instead of copying an older
|
||||||
|
canonical file into place or editing the manifest. See
|
||||||
|
[Troubleshooting: Analysis artifact evidence is not current](./troubleshooting.md#analysis-artifact-evidence-is-not-current).
|
||||||
|
|
||||||
|
## Artifact Selection
|
||||||
|
|
||||||
|
For a configured artifact family, selecting its family key expands to every
|
||||||
|
concrete character artifact. Select a concrete generated key to operate on one
|
||||||
|
member only. Manifests and plan output retain the concrete key as the durable
|
||||||
|
identity and include the family and character ID as optional provenance.
|
||||||
|
|
||||||
|
`--artifacts` can be used on `run`, `session plan`, `run-stage`, `analyze`, and
|
||||||
|
`publish`. For a bounded run or plan, the selected range must contain `analyze`
|
||||||
|
or `publish`.
|
||||||
|
|
||||||
|
Selection behavior:
|
||||||
|
|
||||||
|
- validates names against `pipeline.scriptorium.artifacts`;
|
||||||
|
- selects explicit analyze targets and permits their required configured
|
||||||
|
prerequisites to be reused or rebuilt first;
|
||||||
|
- filters publish rules for `narratio.artifact.<name>` sources only;
|
||||||
|
- does not suppress built-in transcript, bounds, or explicitly configured
|
||||||
|
`narratio.extraction.<name>` publish sources; and
|
||||||
|
- never partially selects Notarius lanes.
|
||||||
|
|
||||||
|
## Extraction Workflow
|
||||||
|
|
||||||
|
When Notarius is omitted or disabled, `extract` records an explicit skipped
|
||||||
|
outcome with reason `notarius_disabled` and no outputs. A later invocation
|
||||||
|
reconsiders the skipped stage, so enabling Notarius does not require force.
|
||||||
|
|
||||||
|
When Notarius extraction is enabled, the stage consumes the final trimmed JSON
|
||||||
|
and preserves the complete validated Notarius bundle at:
|
||||||
|
|
||||||
|
- `artifacts/notarius/{narratio_run_id}/`
|
||||||
|
|
||||||
|
The directory is immutable once promoted. Configured lanes become
|
||||||
|
`narratio.extraction.<name>` sources for Scriptorium and explicit publish rules;
|
||||||
|
the bundle and `index.json` are retained for audit and resume validation but
|
||||||
|
are not selectable or published implicitly.
|
||||||
|
|
||||||
|
Configured Notarius references resolve only from the current manifest-backed
|
||||||
|
prepared inputs. Their canonical locations are `inputs/party.yml`,
|
||||||
|
`inputs/players.yml`, `inputs/glossary.yml`, and, when configured,
|
||||||
|
`inputs/spell_catalog.json`. Extraction supplies Notarius with verified copies
|
||||||
|
under `runs/<run_id>/extract/references/` so a concurrent refresh of canonical
|
||||||
|
prepared files cannot change the bytes consumed by an in-flight invocation.
|
||||||
|
For a canonical party, preparation retains the validated authored party bytes
|
||||||
|
at `inputs/party.yml` and generates `inputs/players.yml` from that roster.
|
||||||
|
The manifest records their checksums separately, with the players input marked
|
||||||
|
as derived from the party; refresh preparation after changing the roster rather
|
||||||
|
than editing either prepared file.
|
||||||
|
Inspect the effective stable-input inventory and
|
||||||
|
prepared-file readiness with:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session status 2026-04-04
|
||||||
|
narratio session validate 2026-04-04
|
||||||
|
```
|
||||||
|
|
||||||
|
Reference metadata records selector, source ID, session-relative path,
|
||||||
|
checksum, and byte size, but never payload contents. Changing a prepared
|
||||||
|
reference changes extraction identity: ordinary continuation rejects the old
|
||||||
|
result, reruns Notarius, and marks successful downstream stages stale. If the
|
||||||
|
prepared file is missing or inconsistent with its manifest checksum, repair
|
||||||
|
the source configuration and refresh prepared state first:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run-stage prepare 2026-04-04 --force
|
||||||
|
```
|
||||||
|
|
||||||
|
Starting a replacement clears the previous extraction payload from the current
|
||||||
|
session-stage record. If that replacement fails or self-skips, the current
|
||||||
|
record does not fall back to the earlier outputs. The earlier run manifest and
|
||||||
|
immutable bundle remain available for inspection, but downstream resolution
|
||||||
|
requires a new current successful extraction record.
|
||||||
|
|
||||||
|
Atomic Notarius bundle promotion is supported on Linux and macOS. On Windows
|
||||||
|
and other operating systems, extraction fails before copying the bundle into a
|
||||||
|
temporary promotion tree because Narratio has no verified atomic no-replace
|
||||||
|
directory primitive there. This is an extraction limitation, not a broader
|
||||||
|
platform-support guarantee for every Narratio workflow.
|
||||||
|
|
||||||
|
## External Command Lifecycle
|
||||||
|
|
||||||
|
When an external command is cancelled or times out, Narratio terminates its
|
||||||
|
owned descendants as well as the command itself. Cancellation first requests
|
||||||
|
termination where the platform supports it, then force terminates after a
|
||||||
|
bounded wait. A command is not considered finished until its leader has been
|
||||||
|
reaped, and descendants that keep standard output or error open cannot keep
|
||||||
|
the invocation blocked. Other operating systems fail closed rather than launch
|
||||||
|
a command without tree ownership.
|
||||||
|
|
||||||
|
Subprocess stdout and stderr diagnostics are separately redacted and capped at
|
||||||
|
8 MiB per invocation. Narratio does not retain configured credential values in
|
||||||
|
these logs or their error tails; reaching a capture limit terminates the command
|
||||||
|
tree and reports which stream exceeded the limit.
|
||||||
|
|
||||||
|
Run-local diagnostics are:
|
||||||
|
|
||||||
|
- `runs/{run_id}/extract/notarius.receipt.json`
|
||||||
|
- `runs/{run_id}/extract/notarius.stderr.log`
|
||||||
|
- `runs/{run_id}/extract/notarius-output/` before durable promotion
|
||||||
|
|
||||||
|
The run-record upload is an allowlist derived from the validated run manifest,
|
||||||
|
not a workspace scan. Each declared source is opened without following
|
||||||
|
symlinked ancestors or the leaf, verified as a regular file, and streamed from
|
||||||
|
that verified descriptor. Unlisted files and unsafe entries are never uploaded.
|
||||||
|
The durable bundle is never scanned for implicit publication; only lanes named
|
||||||
|
by explicit `pipeline.publish.outputs` rules are uploaded.
|
||||||
|
|
||||||
|
To intentionally replace the current extraction result, run:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run-stage extract 2026-04-04 --force
|
||||||
|
```
|
||||||
|
|
||||||
|
Narratio automatically reruns extraction when its recorded invocation contract,
|
||||||
|
prepared Narratio reference identities, or durable output validation changes.
|
||||||
|
The semantic portion covers Notarius enablement, pipeline identity, declared
|
||||||
|
reference mapping, and output contracts. Executable, timeout, working directory,
|
||||||
|
and private config-file paths are operational and do not invalidate a current
|
||||||
|
result.
|
||||||
|
It cannot fingerprint configuration files, profiles, prompts, modules, or
|
||||||
|
other references loaded transitively by Notarius itself. Force extraction after
|
||||||
|
changing any of those inputs, even when the top-level Narratio and Notarius
|
||||||
|
config paths remain the same. A forced extract
|
||||||
|
marks successful downstream stages stale. Ordinary extraction failures or
|
||||||
|
outcome changes also stale affected downstream stages, while an identical
|
||||||
|
repeated `notarius_disabled` self-skip does not repeatedly invalidate them.
|
||||||
|
|
||||||
|
Publish reuse additionally tracks enabled/run-upload behavior, normalized
|
||||||
|
output rules, static locks, and remote backend/bucket/region/endpoint/root
|
||||||
|
identity. Credential environment names, local workspace placement, and run IDs
|
||||||
|
are excluded. Regardless of semantic reuse evidence, executing publish still
|
||||||
|
revalidates mutable remote locks immediately before commit selection.
|
||||||
|
|
||||||
|
## Publish Workflow
|
||||||
|
|
||||||
|
Run publish only:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio publish 2026-04-04
|
||||||
|
```
|
||||||
|
|
||||||
|
Equivalent:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run-stage publish 2026-04-04 --force
|
||||||
|
```
|
||||||
|
|
||||||
|
Publish commit model:
|
||||||
|
|
||||||
|
- uploads eligible run files under `{session_prefix}/runs/{run_id}/`, excluding
|
||||||
|
audio and the run-local Notarius staging bundle;
|
||||||
|
- uploads configured published outputs and `previous/**` cache files into the
|
||||||
|
same immutable run scope, including only explicitly configured extraction
|
||||||
|
lanes;
|
||||||
|
- writes `{session_prefix}/runs/{run_id}/commit.json` after all declared
|
||||||
|
immutable objects are uploaded and verified; and
|
||||||
|
- writes `{session_prefix}/current/commit-pointer.json` once, last.
|
||||||
|
|
||||||
|
`current/commit-pointer.json` is the remote current-state commit marker. It
|
||||||
|
selects exactly one immutable commit, which declares the complete object set.
|
||||||
|
|
||||||
|
## Remote Commit Migration
|
||||||
|
|
||||||
|
The immutable remote commit contract uses
|
||||||
|
`runs/{run_id}/commit.json` to declare a run's complete object set and a small
|
||||||
|
`current/commit-pointer.json` to select it. The pointer binds the selected
|
||||||
|
commit by version, checksum, size, and storage generation; committed artifacts
|
||||||
|
are also checksum- and generation-bound. Readers accept this contract now and
|
||||||
|
strictly reject mismatched or unknown data.
|
||||||
|
|
||||||
|
Legacy reads are limited to a coherent `current/manifest.json` and
|
||||||
|
`current/run_id.txt` pair; a torn pair is rejected. New publication does not
|
||||||
|
write that pair and remote commit state does not carry local
|
||||||
|
`current_pointer_written` metadata.
|
||||||
|
|
||||||
|
## Publish Locks
|
||||||
|
|
||||||
|
Lock sources:
|
||||||
|
|
||||||
|
- static locks in `pipeline.publish.locks`
|
||||||
|
- mutable remote locks in `{session_prefix}/locks.yml`
|
||||||
|
|
||||||
|
Effective lock rules:
|
||||||
|
|
||||||
|
- static and remote locks are merged;
|
||||||
|
- static locks win on source collisions;
|
||||||
|
- locked outputs are intentional skips;
|
||||||
|
- lock add/remove commands mutate only remote lock state through generation-bound
|
||||||
|
conditional writes. A command retries a bounded number of concurrent
|
||||||
|
conflicts while its invocation context remains active, so it never replaces a
|
||||||
|
different lock-document generation; and
|
||||||
|
- a publish re-reads remote locks immediately before it writes the current
|
||||||
|
commit pointer. A lock committed before that recheck prevents selecting the
|
||||||
|
new snapshot, even though its already-uploaded immutable objects may remain
|
||||||
|
available for a later retry.
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session locks 2026-04-04
|
||||||
|
narratio session locks add 2026-04-04 narratio.artifact.session_recap --reason "manual edits" --force
|
||||||
|
narratio session locks remove 2026-04-04 narratio.artifact.session_recap
|
||||||
|
```
|
||||||
|
|
||||||
|
## Restore Workflow
|
||||||
|
|
||||||
|
Use restore when local durable session state is missing or stale and remote committed current state is authoritative.
|
||||||
|
|
||||||
|
Dry run:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session restore 2026-04-04 --dry-run
|
||||||
|
```
|
||||||
|
|
||||||
|
Apply:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session restore 2026-04-04
|
||||||
|
```
|
||||||
|
|
||||||
|
`--dry-run` does not write durable session files. It still reads the selected
|
||||||
|
remote current state and may read object identity/content needed to classify the
|
||||||
|
plan, so it is not a network-free operation.
|
||||||
|
|
||||||
|
Default restore scope:
|
||||||
|
|
||||||
|
- the committed session manifest and the committed transcript/artifact objects
|
||||||
|
declared by the selected remote commit
|
||||||
|
- `previous/**` when needed by configured previous-session artifact inputs
|
||||||
|
|
||||||
|
Optional:
|
||||||
|
|
||||||
|
- `--include-audio` to include `audio/**`
|
||||||
|
- `--force` to overwrite eligible conflicting regular files; it never replaces
|
||||||
|
directories or other non-regular local targets
|
||||||
|
|
||||||
|
Restore writes an execution report at `reports/restore-latest.json`.
|
||||||
|
|
||||||
|
If restore fails after beginning installation, it leaves a durable
|
||||||
|
`.restore-incomplete.json` marker in the session root. Pipeline runs will stop
|
||||||
|
until you rerun the same restore command and it completes. Restore intentionally
|
||||||
|
does not try to roll back files already installed; retrying the selected remote
|
||||||
|
snapshot is the recovery procedure.
|
||||||
|
|
||||||
|
## Local State Layout
|
||||||
|
|
||||||
Session root:
|
Session root:
|
||||||
- `{workspace.root}/work/{campaign}/{session_id}/`
|
|
||||||
|
|
||||||
Primary state:
|
- `{workspace.root}/work/{campaign}/{session_id}`
|
||||||
- `manifest.json`: session-level stage state.
|
|
||||||
- `runs/{run_id}/manifest.json`: invocation-level state.
|
|
||||||
- `.lock`: session lock while a run is active.
|
|
||||||
|
|
||||||
Canonical session directories:
|
Durable session paths:
|
||||||
- `inputs/`
|
|
||||||
- `audio/`
|
|
||||||
- `transcripts/`
|
|
||||||
- `artifacts/`
|
|
||||||
- `reports/`
|
|
||||||
- `logs/`
|
|
||||||
- `config/`
|
|
||||||
- `current/`
|
|
||||||
- `runs/`
|
|
||||||
|
|
||||||
Run-local stage directories:
|
- `manifest.json`
|
||||||
- `runs/{run_id}/{stage}/` with stage-local `outputs/`, `logs/`, `reports/`, `config/`, `scratch/`.
|
- `inputs/**`
|
||||||
|
- `audio/**`
|
||||||
|
- `transcripts/**`
|
||||||
|
- `artifacts/**`
|
||||||
|
- `previous/**`
|
||||||
|
- `reports/**`
|
||||||
|
- `logs/**`
|
||||||
|
- `config/**`
|
||||||
|
- `runs/**`
|
||||||
|
|
||||||
Behavior:
|
Validated Notarius bundles live below `artifacts/notarius/{run_id}/`; receipt,
|
||||||
- directory creation is idempotent.
|
stderr, and pre-promotion output remain in the producing run's `extract`
|
||||||
- stage outputs are generally generated run-local first, then promoted to canonical paths on success.
|
directory as described in [Extraction Workflow](#extraction-workflow).
|
||||||
|
|
||||||
## Analyze artifact execution lifecycle
|
Run-local layout:
|
||||||
|
|
||||||
Analyze executes configured artifacts from `pipeline.scriptorium.artifacts`.
|
- `runs/{run_id}/{stage}/outputs`
|
||||||
|
- `runs/{run_id}/{stage}/logs`
|
||||||
|
- `runs/{run_id}/{stage}/reports`
|
||||||
|
- `runs/{run_id}/{stage}/config`
|
||||||
|
- `runs/{run_id}/{stage}/scratch`
|
||||||
|
|
||||||
Execution model:
|
Spool layout (runtime/transient):
|
||||||
- executable set = enabled artifacts, filtered by `--artifacts` when provided.
|
|
||||||
- artifact-to-artifact dependencies are declared via `depends_on`.
|
|
||||||
- selected artifacts run in deterministic dependency order.
|
|
||||||
- after each successful artifact run, output is promoted to configured canonical `output_path`.
|
|
||||||
|
|
||||||
Configured artifact source reuse:
|
- `{spool.root}/{campaign}/{session_id}/{run_id}/...`
|
||||||
- a non-executable configured artifact can satisfy inputs if its configured output file already exists and is valid.
|
- restore audio spool under `{spool.root}/{campaign}/{session_id}/restore/audio`
|
||||||
- reused configured artifact provenance is `filesystem.disabled_artifact_output`.
|
|
||||||
|
|
||||||
`--artifacts` behavior:
|
Cache layout (durable S3 audio cache):
|
||||||
- accepted on `run`, `resume`, and `run-stage analyze`.
|
|
||||||
- filters analyze execution only; does not force stage rerun.
|
|
||||||
|
|
||||||
## Remote archive layout and publish contract
|
- `{cache.root}/s3/{bucket}/...`
|
||||||
|
|
||||||
When archive is enabled and run upload is enabled, archive publishes under:
|
Each cached audio file has an adjacent managed identity record. It binds the
|
||||||
|
file to its remote object version and verified digest; deleting or altering the
|
||||||
|
record simply causes Narratio to download and verify the object again.
|
||||||
|
|
||||||
- session prefix: `{root_prefix}/campaigns/{campaign}/sessions/{session_id}/`
|
### Workspace Permissions
|
||||||
- run prefix: `{session_prefix}/runs/{run_id}/`
|
|
||||||
|
|
||||||
Archive uploads:
|
Ordinary Narratio workspace content is intentionally shareable with the
|
||||||
- run record files from run root (excluding `audio/`).
|
workspace group. On POSIX systems, Narratio-created workspace, spool, and cache
|
||||||
- promoted files from explicit `archive.promote_artifacts` rules.
|
directories converge on setgid `02775`; ordinary files, including manifests,
|
||||||
|
transcripts, generated configuration, logs, reports, and Notarius artifacts,
|
||||||
|
converge on `0664`. Narratio explicitly applies these modes so a restrictive
|
||||||
|
caller umask does not remove group write or setgid. It does not change file or
|
||||||
|
directory ownership: the configured workspace's existing group is inherited.
|
||||||
|
|
||||||
Publish order:
|
Windows does not implement POSIX mode bits or setgid semantics. Configure the
|
||||||
1. upload `current/manifest.json`
|
workspace, spool, and cache locations with an ACL that grants the collaborating
|
||||||
2. upload `current/run_id.txt` last
|
group read/write access, and configure credential locations with an ACL limited
|
||||||
|
to the intended credential owner. Do not use POSIX mode displays as evidence of
|
||||||
|
Windows access control.
|
||||||
|
|
||||||
`current/run_id.txt` is the remote commit marker.
|
API keys are credentials, not ordinary workspace data. Store them outside the
|
||||||
|
shared workspace or in a separately restricted credential location; ordinary
|
||||||
|
workspace group access must never be treated as authorization to read keys.
|
||||||
|
On POSIX, provision a credential directory as `0700` and credential files as
|
||||||
|
`0600`; Narratio rejects group- or other-readable configured credential paths.
|
||||||
|
On Windows, restrict the directory and files with ACLs to the credential owner.
|
||||||
|
|
||||||
Archive promotion is explicit and path-based:
|
External adapter results are individually bounded before Narratio validates or
|
||||||
- Narratio does not auto-promote all generated analyze artifacts.
|
materializes them. These per-file limits do not reserve disk space: prevent hard
|
||||||
- missing required promotion sources fail archive stage.
|
disk exhaustion with filesystem, service, container, or volume quotas sized for
|
||||||
- missing optional promotion sources are skipped.
|
the session workload.
|
||||||
|
|
||||||
## Resume, retry, and safe rerun behavior
|
## Cleanup
|
||||||
|
|
||||||
Default skip:
|
Session-scoped cleanup:
|
||||||
- `run` and `run-stage` skip already-succeeded stages unless `--force` is set.
|
|
||||||
|
|
||||||
Resume:
|
|
||||||
- `resume` starts at first non-succeeded stage.
|
|
||||||
- `resume --force` runs full stage order.
|
|
||||||
|
|
||||||
Forced reruns:
|
|
||||||
- force-rerunning an upstream succeeded stage marks downstream succeeded stages as `stale`.
|
|
||||||
|
|
||||||
Safe rerun pattern:
|
|
||||||
1. rerun the changed stage with `--force`.
|
|
||||||
2. run `resume` to rebuild downstream stages.
|
|
||||||
|
|
||||||
## Cleanup behavior
|
|
||||||
|
|
||||||
Cleanup is considered only when archive stage executed and succeeded.
|
|
||||||
|
|
||||||
Cleanup toggles:
|
|
||||||
- `pipeline.spool.delete_audio_after_archive=true` deletes run-scoped spool audio.
|
|
||||||
- `pipeline.workspace.cleanup_after_archive=true` deletes run-scoped local run directory.
|
|
||||||
|
|
||||||
Cleanup eligibility gates:
|
|
||||||
- archive enabled
|
|
||||||
- archive run upload enabled
|
|
||||||
- run record upload completed
|
|
||||||
- current pointer write completed (`current/run_id.txt` written)
|
|
||||||
|
|
||||||
No cleanup for failed/incomplete/unarchived/archive-skipped runs.
|
|
||||||
|
|
||||||
## Failure and recovery playbooks
|
|
||||||
|
|
||||||
After failure, Narratio keeps:
|
|
||||||
- session manifest
|
|
||||||
- run manifest
|
|
||||||
- run-local artifacts/logs/config/reports
|
|
||||||
|
|
||||||
Failed or incomplete runs remain local-only.
|
|
||||||
|
|
||||||
Recommended recovery:
|
|
||||||
|
|
||||||
1. inspect state:
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio status --manifest <manifest-path>
|
narratio clean 2026-04-04
|
||||||
```
|
```
|
||||||
|
|
||||||
2. fix root cause (config/input/credentials/service availability).
|
Global cleanup:
|
||||||
3. continue with `resume`, or targeted `run-stage --force` followed by `resume`.
|
|
||||||
|
|
||||||
## Operational caveats
|
```bash
|
||||||
|
narratio clean --all
|
||||||
|
```
|
||||||
|
|
||||||
- `status` requires explicit `--manifest`; there is no session-id lookup command.
|
Dry-run and cache variants:
|
||||||
- local and S3 audio input modes are mutually exclusive.
|
|
||||||
- archive publish requires upstream stages through `analyze` to be `succeeded`.
|
```bash
|
||||||
- required promotion rules can fail when selected analyze artifacts did not generate a required file path.
|
narratio clean 2026-04-04 --dry-run --clear-cache
|
||||||
|
narratio clean --all --dry-run --clear-cache
|
||||||
|
```
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
|
||||||
|
- `clean` deletes work/spool session state;
|
||||||
|
- cache is preserved unless `--clear-cache` is set;
|
||||||
|
- each deletion is confined beneath its configured workspace, spool, or cache
|
||||||
|
root and refuses symlinked paths;
|
||||||
|
- automatic post-publish cleanup is gated by successful publish commit plus:
|
||||||
|
- `pipeline.spool.delete_audio_after_publish=true`
|
||||||
|
- `pipeline.workspace.cleanup_after_publish=true`
|
||||||
|
- Narratio first records the exact run-scoped cleanup obligation. If cleanup
|
||||||
|
reports incomplete, the remote committed snapshot remains current; rerun
|
||||||
|
publish to retry only the outstanding confined local cleanup.
|
||||||
|
|
||||||
|
Post-publish cleanup is evaluated only when `publish` actually executes in the
|
||||||
|
current invocation. A bounded range that excludes publish does not replay a
|
||||||
|
cleanup obligation as an unrelated side effect.
|
||||||
|
|
||||||
|
## Operational Caveats
|
||||||
|
|
||||||
|
- Local and S3 audio modes are mutually exclusive.
|
||||||
|
- Publish requires prerequisite stages through `render` and `analyze` to be succeeded.
|
||||||
|
- Markdown publish defaults require render outputs (`transcripts/final.md` and `transcripts/final.trimmed.md`).
|
||||||
|
- Restore requires configured object storage and committed remote current state.
|
||||||
|
- Storage-backed commands load filesystem secrets before object-store initialization.
|
||||||
|
|||||||
286
docs/policy/architecture.md
Normal file
286
docs/policy/architecture.md
Normal file
@@ -0,0 +1,286 @@
|
|||||||
|
# Architecture
|
||||||
|
|
||||||
|
This document defines Narratio's intended high-level architecture and the
|
||||||
|
invariants that changes must preserve. Implemented component details belong in
|
||||||
|
the [Internal Overview](../internal/overview.md) and its linked documents.
|
||||||
|
Significant architectural decision history belongs under `docs/adr/` when such
|
||||||
|
records exist.
|
||||||
|
|
||||||
|
## System Shape
|
||||||
|
|
||||||
|
Narratio is a small Go application that turns D&D session audio into polished
|
||||||
|
transcripts and generated session artifacts. It is an explicit, stage-driven
|
||||||
|
orchestrator, not a general workflow engine.
|
||||||
|
|
||||||
|
Narratio coordinates specialized external systems rather than reimplementing
|
||||||
|
their domains:
|
||||||
|
|
||||||
|
- WhisperX performs transcription;
|
||||||
|
- Seriatim performs deterministic transcript processing and rendering;
|
||||||
|
- Audita performs transcript correction and polishing;
|
||||||
|
- Notarius extracts validated structured artifact bundles; and
|
||||||
|
- Scriptorium executes prompts and produces configured artifacts.
|
||||||
|
|
||||||
|
Narratio owns orchestration, configuration resolution, session and run state,
|
||||||
|
artifact and path modeling, manifest persistence, stage sequencing, resume,
|
||||||
|
restore, cleanup gates, and publish semantics. External contracts are defined
|
||||||
|
in the [integration documentation](../integrations/).
|
||||||
|
|
||||||
|
The pipeline has one canonical ordered stage set. Configuration may enable,
|
||||||
|
disable, or parameterize supported behavior, but it must not turn that sequence
|
||||||
|
into an arbitrary DAG or hide orchestration in generic workflow abstractions.
|
||||||
|
An invocation selects either the full sequence or one inclusive contiguous
|
||||||
|
range of it. Execution remains flat and canonical even though invalidation is
|
||||||
|
dependency-aware: the application owns a separate fixed relation used only to
|
||||||
|
stale transitive dependents, including dependents outside a selected range.
|
||||||
|
The implemented stage inventory belongs in the
|
||||||
|
[Internal Overview](../internal/overview.md).
|
||||||
|
|
||||||
|
Narratio is contract-first without being abstraction-heavy. Interfaces and
|
||||||
|
extension points should protect demonstrated boundaries. New abstraction is not
|
||||||
|
itself an architectural goal.
|
||||||
|
|
||||||
|
## Ownership And Dependency Direction
|
||||||
|
|
||||||
|
The application boundary owns command dispatch, configuration selection,
|
||||||
|
production composition, session locking, and top-level lifecycle. It may depend
|
||||||
|
on concrete implementations to assemble a run.
|
||||||
|
|
||||||
|
Stage orchestration expresses intent in Narratio-level data and interfaces.
|
||||||
|
Stages may depend on configuration, manifest, artifact, path, and adapter
|
||||||
|
contracts, but they must not depend on transport-specific request types,
|
||||||
|
subprocess argument construction, cloud SDK types, or downstream tool internals.
|
||||||
|
|
||||||
|
Adapters translate between Narratio contracts and external systems. They own
|
||||||
|
HTTP, subprocess, notification, and object-storage mechanics, including command
|
||||||
|
construction, transport behavior, provider response handling, and external
|
||||||
|
error adaptation. External dependency types must remain inside the adapter that
|
||||||
|
owns them unless that dependency is the adapter's explicit public contract.
|
||||||
|
WhisperX HTTP behavior, Seriatim, Audita, Notarius, and Scriptorium command
|
||||||
|
construction, notification transport, and object-storage SDK details remain
|
||||||
|
behind these boundaries.
|
||||||
|
|
||||||
|
State and path services must not infer stage policy. Storage implementations
|
||||||
|
receive explicit bucket-relative keys and do not infer campaign, session, run,
|
||||||
|
or root-prefix semantics. Manifest persistence records transitions but does not
|
||||||
|
choose orchestration policy. Artifact resolution identifies and validates
|
||||||
|
artifacts but does not execute producers.
|
||||||
|
|
||||||
|
Dependencies should remain narrow and point toward Narratio-owned contracts.
|
||||||
|
Prefer the Go standard library. Add an external dependency only when it provides
|
||||||
|
a clear correctness, security, interoperability, or complexity benefit, and
|
||||||
|
confine it to the boundary that needs it.
|
||||||
|
|
||||||
|
## Stage Boundaries
|
||||||
|
|
||||||
|
Each stage has one explicit responsibility and declares:
|
||||||
|
|
||||||
|
- required input state;
|
||||||
|
- produced output state;
|
||||||
|
- configuration it consumes;
|
||||||
|
- external adapters it uses;
|
||||||
|
- manifest references and metadata it reads or writes;
|
||||||
|
- skip, force, invalidation, and resume behavior; and
|
||||||
|
- failure behavior.
|
||||||
|
|
||||||
|
Stages write and validate run-local results before materializing canonical
|
||||||
|
outputs where that distinction applies. A stage is complete only after its
|
||||||
|
required outputs have been written, validated, and recorded in durable manifest
|
||||||
|
state. Later stages depend on recorded success and artifact resolution, not
|
||||||
|
merely on incidental files existing on disk.
|
||||||
|
|
||||||
|
A failed or interrupted stage must not be presented as successful. Failure
|
||||||
|
should preserve enough local state and diagnostics for inspection, recovery,
|
||||||
|
and resume. Forcing a stage invalidates succeeded transitive dependents
|
||||||
|
according to a fixed application-owned relation that is separate from canonical
|
||||||
|
execution order. The relation is validated against the stage inventory and is
|
||||||
|
not configurable.
|
||||||
|
|
||||||
|
A stage may explicitly self-skip with a stable reason and no outputs. That
|
||||||
|
outcome is persisted, clears older outputs owned by the stage, and is
|
||||||
|
reconsidered on a later invocation. A stage may also validate whether an
|
||||||
|
otherwise successful recorded result is still resumable; an obsolete result
|
||||||
|
is staled and rerun, while an unsafe condition that prevents a sound decision
|
||||||
|
stops execution.
|
||||||
|
|
||||||
|
Shared behavior should live behind a narrow service or helper with one clear
|
||||||
|
owner. Stages must not reach across boundaries or reproduce adapter, manifest,
|
||||||
|
artifact, or path policy ad hoc.
|
||||||
|
|
||||||
|
## Manifest, Resume, And Restore
|
||||||
|
|
||||||
|
The session manifest is the durable ledger for progress across invocations. It
|
||||||
|
records session and run identity, stage state, input and output references,
|
||||||
|
diagnostic references, checksums or provenance where useful, and non-secret
|
||||||
|
adapter and publish metadata.
|
||||||
|
|
||||||
|
Resume and skip decisions are manifest-driven. Filesystem state may be
|
||||||
|
inspected and validated, but file presence alone does not replace recorded
|
||||||
|
stage state. Invocation-scoped run records provide an audit of one execution;
|
||||||
|
they do not replace the session manifest as progress authority.
|
||||||
|
|
||||||
|
Restore treats committed remote current state as its authority. It must plan
|
||||||
|
deterministically, confine remote-to-local paths, protect local conflicts, and
|
||||||
|
install the validated session manifest after other restored durable files. The
|
||||||
|
physical workflow and recovery procedures belong in
|
||||||
|
[Operations](../operations.md).
|
||||||
|
|
||||||
|
Restore and runner transitions for one session use the same local lock. A
|
||||||
|
durable incomplete-restore marker blocks runner reuse after a partial restore;
|
||||||
|
safe retry, rather than rollback of arbitrary local effects, is the recovery
|
||||||
|
mechanism. Restored manifest-local references must be confined to the selected
|
||||||
|
local session root, never trusted as producer-machine absolute paths.
|
||||||
|
|
||||||
|
For the immutable remote-commit protocol, a restore or status operation binds
|
||||||
|
to one pointer-selected commit and only its declared object identities. A force
|
||||||
|
flag may replace an eligible regular managed file, but never turns a directory
|
||||||
|
or other non-regular conflict into a successful restore.
|
||||||
|
|
||||||
|
## Configuration
|
||||||
|
|
||||||
|
Configuration is strict, explicit, centralized, and operator-oriented.
|
||||||
|
|
||||||
|
- YAML decoding rejects unknown fields.
|
||||||
|
- Defaults are centralized and testable.
|
||||||
|
- Empty configured values do not silently replace meaningful defaults.
|
||||||
|
- Validation rejects invalid composition before stage execution where
|
||||||
|
practical.
|
||||||
|
- Root-owned imports and one selected profile resolve deterministically through
|
||||||
|
the configuration owner; commands do not implement their own merge rules.
|
||||||
|
- Canonical party rosters are campaign-owned. Their derived players projection
|
||||||
|
and concrete character-family artifacts are resolved before runtime stages
|
||||||
|
or adapters receive configuration.
|
||||||
|
- Resume uses stage- or artifact-owned semantic evidence for observable
|
||||||
|
result-affecting configuration; profile identity and an effective digest are
|
||||||
|
provenance, never blanket cache keys.
|
||||||
|
- Session templating remains narrow and deterministic rather than becoming a
|
||||||
|
general configuration language.
|
||||||
|
- Secret values are supplied indirectly and are not persisted in ordinary
|
||||||
|
configuration.
|
||||||
|
|
||||||
|
Narratio must not become a second configuration system for downstream tools.
|
||||||
|
External systems own their runtime defaults wherever practical; Narratio passes
|
||||||
|
the paths required by its stage contracts and explicit operator overrides. The
|
||||||
|
field-level contract and credential-supply mechanisms belong in
|
||||||
|
[Configuration](../config.md).
|
||||||
|
|
||||||
|
## Artifacts, Paths, And Storage
|
||||||
|
|
||||||
|
Artifact identities and local and remote paths are application contracts.
|
||||||
|
Canonical helpers own workspace, spool, cache, session, run, input, transcript,
|
||||||
|
artifact, log, report, configuration, and publish-current paths. Callers must
|
||||||
|
not reconstruct canonical paths through scattered string concatenation.
|
||||||
|
|
||||||
|
Reusable audio cache entries require a typed record that binds a confined,
|
||||||
|
no-follow regular file and its digest to the selected remote object identity.
|
||||||
|
Size alone and unqualified multipart ETags are not content-integrity evidence.
|
||||||
|
|
||||||
|
Artifact resolution is deterministic and manifest-aware. Producers materialize
|
||||||
|
canonical outputs before reporting success, and consumers resolve declared
|
||||||
|
artifact identities rather than infer files from unrelated directory contents.
|
||||||
|
External artifact bundles become current only through validated immutable
|
||||||
|
promotion and manifest records; directory presence alone never establishes
|
||||||
|
availability.
|
||||||
|
|
||||||
|
Writes, moves, replacements, and deletions must use narrow, explicit,
|
||||||
|
root-confined destinations. Symlinks, traversal, broad roots, and ambiguous
|
||||||
|
relative destinations must not expand the scope of an operation. Cleanup is
|
||||||
|
permitted only through explicit operator action or configured post-publish
|
||||||
|
gates, and it must preserve durable cache unless cache removal is explicitly
|
||||||
|
requested.
|
||||||
|
|
||||||
|
Physical layout, retention, and operational lifecycle belong in
|
||||||
|
[Operations](../operations.md). Logical external formats and durable integration
|
||||||
|
contracts belong under [Integrations](../integrations/).
|
||||||
|
|
||||||
|
## Publish Commit Boundary
|
||||||
|
|
||||||
|
Publish has one explicit remote commit boundary. A remote run becomes current
|
||||||
|
only after Narratio has successfully uploaded its immutable run-scoped objects,
|
||||||
|
the immutable commit manifest, and finally the current commit pointer.
|
||||||
|
|
||||||
|
`current/commit-pointer.json` is the sole mutable selector and must be written
|
||||||
|
exactly once, last. Failed, incomplete, skipped, or uncommitted publish attempts
|
||||||
|
must not be presented as current remote state. Publish locks remain authoritative
|
||||||
|
and are not bypassed by a forced run. Mutable remote locks use provider-enforced
|
||||||
|
generation preconditions and are revalidated immediately before pointer
|
||||||
|
selection; loss of that check leaves the prior committed snapshot current.
|
||||||
|
|
||||||
|
Automatic local cleanup is permitted only after a successful publish commit,
|
||||||
|
only when explicitly configured, and only through the path-safety guardrails.
|
||||||
|
It is a durable local obligation bound to that committed run and its exact
|
||||||
|
targets, not an inferred side effect of the current stage list. A cleanup
|
||||||
|
failure makes the invocation incomplete while leaving the committed remote
|
||||||
|
snapshot authoritative; later invocations resume the recorded obligation.
|
||||||
|
|
||||||
|
## Security, Privacy, And Diagnostics
|
||||||
|
|
||||||
|
Narratio distinguishes ordinary workspace data from credentials. Campaign and
|
||||||
|
session material—including manifests, transcripts, prompts, generated
|
||||||
|
configuration, logs, reports, diagnostics, and Notarius artifacts—is
|
||||||
|
intentionally shareable with the configured workspace group. API-key material
|
||||||
|
is sensitive and is not covered by the ordinary workspace-sharing policy.
|
||||||
|
|
||||||
|
On POSIX systems, Narratio-created ordinary workspace directories converge on
|
||||||
|
setgid `02775` and ordinary workspace files on `0664`, even when the caller's
|
||||||
|
umask is restrictive. This preserves the existing workspace group for nested
|
||||||
|
creation and atomic replacements without changing ownership. API-key storage
|
||||||
|
uses a separate restrictive contract. On Windows, POSIX mode bits and setgid
|
||||||
|
are not authoritative; operators must provide the equivalent shared-group and
|
||||||
|
credential-restricted ACLs described in [Operations](../operations.md#workspace-permissions).
|
||||||
|
|
||||||
|
Raw secrets must not be stored in pipeline, campaign, or session YAML or written
|
||||||
|
to manifests, logs, generated configuration, reports, publish metadata,
|
||||||
|
documentation, or examples. Secrets enter through configured environment
|
||||||
|
variable names or secret-file references. Diagnostics should avoid transcript
|
||||||
|
and prompt content unless a deliberate, bounded inspection mechanism requires
|
||||||
|
it.
|
||||||
|
|
||||||
|
Logs, reports, generated invocation files, generated configuration, and render
|
||||||
|
debug files are diagnostics, not canonical pipeline products. They should be
|
||||||
|
durable and discoverable where configured, and manifest references must preserve
|
||||||
|
the distinction between diagnostics and artifacts.
|
||||||
|
|
||||||
|
Documentation security rules belong in the
|
||||||
|
[Documentation Policy](documentation.md). Credential supply belongs in
|
||||||
|
[Configuration](../config.md), while permissions, sensitive runtime-artifact
|
||||||
|
handling, and recovery belong in [Operations](../operations.md).
|
||||||
|
|
||||||
|
## Determinism And Testability
|
||||||
|
|
||||||
|
Narratio prefers deterministic behavior where practical, including stable local
|
||||||
|
and remote layouts, sorted operation order, predictable generated
|
||||||
|
configuration, repeatable command construction, deterministic artifact
|
||||||
|
resolution, and reproducible planning.
|
||||||
|
|
||||||
|
Run IDs and timestamps may be intentionally variable, but surrounding behavior
|
||||||
|
must remain controllable in tests. Core behavior should be testable without live
|
||||||
|
external services; expensive, nondeterministic, destructive, or external
|
||||||
|
boundaries should be replaceable with focused test doubles. General testing
|
||||||
|
philosophy and sufficiency rules belong in the [Testing Policy](testing.md).
|
||||||
|
|
||||||
|
## Documentation And Decision Records
|
||||||
|
|
||||||
|
Documentation follows the [Documentation Policy](documentation.md). Current
|
||||||
|
behavior belongs in its canonical user, operator, integration, architecture, or
|
||||||
|
internal owner. Proposed behavior and implementation status belong under
|
||||||
|
`docs/roadmap/`.
|
||||||
|
|
||||||
|
Significant architectural decisions may be recorded under `docs/adr/` using the
|
||||||
|
format and lifecycle defined by the documentation policy. ADR acceptance does
|
||||||
|
not establish that a decision has been implemented.
|
||||||
|
|
||||||
|
## Architectural Non-Goals
|
||||||
|
|
||||||
|
Narratio does not aim to provide:
|
||||||
|
|
||||||
|
- a generic DAG or workflow engine;
|
||||||
|
- a replacement configuration layer for WhisperX, Seriatim, Audita,
|
||||||
|
Scriptorium, or other downstream tools;
|
||||||
|
- a storage abstraction broader than the needs of this pipeline;
|
||||||
|
- stage logic coupled directly to cloud SDKs, transports, subprocess details,
|
||||||
|
or downstream implementation internals;
|
||||||
|
- raw-secret persistence;
|
||||||
|
- implicit cross-stage behavior that bypasses manifest and artifact contracts;
|
||||||
|
or
|
||||||
|
- a prompt-authoring system.
|
||||||
150
docs/policy/documentation.md
Normal file
150
docs/policy/documentation.md
Normal file
@@ -0,0 +1,150 @@
|
|||||||
|
# Documentation Policy
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
|
||||||
|
This policy assigns each documentation topic to one canonical owner. Its goal is
|
||||||
|
to keep Narratio documentation accurate, concise, discoverable, and resistant
|
||||||
|
to drift for users, operators, developers, integrators, and LLM coding agents.
|
||||||
|
|
||||||
|
## Core Rules
|
||||||
|
|
||||||
|
### One Canonical Owner
|
||||||
|
|
||||||
|
Each authoritative fact belongs in one document. A non-owning document may give
|
||||||
|
a short, stable summary for orientation, but it must link to the canonical owner
|
||||||
|
instead of repeating volatile details.
|
||||||
|
|
||||||
|
Volatile details include commands, flags, configuration fields and defaults,
|
||||||
|
stage or integration keys, schemas, file names, paths, status codes, retry
|
||||||
|
behavior, and runtime guarantees. If readers could reasonably treat a statement
|
||||||
|
as a contract, maintain it only in the owning document.
|
||||||
|
|
||||||
|
### Current And Future Behavior
|
||||||
|
|
||||||
|
Outside `docs/roadmap/`, documentation describes implemented behavior only.
|
||||||
|
Partial features may be described only to their implemented boundary.
|
||||||
|
|
||||||
|
ADRs are the narrow exception: an ADR may record an accepted architectural
|
||||||
|
decision before implementation, but acceptance must not be presented as proof
|
||||||
|
that the behavior exists. The roadmap owns implementation status and sequencing
|
||||||
|
until the decision is implemented. Current architecture, user, operator,
|
||||||
|
integration, and internal documentation are updated when the behavior lands.
|
||||||
|
|
||||||
|
### Audience And Detail
|
||||||
|
|
||||||
|
Write for the document's stated audience and include only the detail needed for
|
||||||
|
its owned topic. User and operator docs should not expose implementation detail.
|
||||||
|
Developer docs should link to user-facing and external contracts rather than
|
||||||
|
restate them.
|
||||||
|
|
||||||
|
### Examples
|
||||||
|
|
||||||
|
Complete copyable files belong in `examples/`. Documentation may use the
|
||||||
|
smallest illustrative snippet needed to explain its owned topic, but should link
|
||||||
|
to maintained examples instead of embedding a second complete copy.
|
||||||
|
|
||||||
|
Examples must be valid, secret-free, and tested where practical. Commands and
|
||||||
|
configuration used in documentation should match the application.
|
||||||
|
|
||||||
|
### Security And Privacy
|
||||||
|
|
||||||
|
Documentation and examples must not contain real credentials, private keys,
|
||||||
|
private environment dumps, sensitive source material, or private infrastructure
|
||||||
|
details unless intentionally public. Document secret-handling mechanisms, not
|
||||||
|
secret values.
|
||||||
|
|
||||||
|
## Canonical Ownership
|
||||||
|
|
||||||
|
| Topic | Canonical owner | Owned content | Content owned elsewhere |
|
||||||
|
| --- | --- | --- | --- |
|
||||||
|
| Product orientation and minimal end-to-end quickstart | `README.md` | What Narratio is, why it is useful, one shortest successful invocation, and links onward. | Complete command reference, configuration reference, operational procedures, implementation detail. |
|
||||||
|
| Contributor entry point | `docs/development.md` | Task-oriented reading guide, minimal contributor orientation, baseline validation commands, and links to canonical docs. | Package inventory, architecture rules, subsystem behavior, detailed change recipes. |
|
||||||
|
| Maintainer release procedure | `docs/release.md` | Version selection, candidate preparation and validation, guarded tag publication, release completion boundary, failure recovery, and optional asynchronous inspection. | Script implementation mechanics, current application contracts, and historical release summaries. |
|
||||||
|
| Historical release summary | `docs/releases/<tag>.md` | Immutable summary, compatibility, upgrade, and changes for one released version. | Current maintainer procedure and current application contract details. |
|
||||||
|
| Current application architecture | `docs/policy/architecture.md` | System shape, normative ownership, dependency direction, architectural boundaries, invariants, safety properties, and non-goals. | Concrete package inventory, implementation mechanics, contributor procedures, decision history, future work. |
|
||||||
|
| Documentation organization | `docs/policy/documentation.md` | Documentation ownership, audience boundaries, maintenance rules, and ADR/document lifecycle. | Application architecture or product behavior. |
|
||||||
|
| Testing policy | `docs/policy/testing.md` | Test philosophy, risk-based sufficiency, test boundaries, doubles, coverage guidance, regression-test policy, and criteria for adding, rewriting, or deleting tests. | Subsystem behavior, application contracts, subsystem-specific test inventories, and implementation plans. |
|
||||||
|
| CLI contract | `docs/cli.md` | Commands, arguments, flags, invocation semantics, output conventions, and exit behavior. | End-to-end operating procedures, configuration field definitions, runtime filesystem layout, stage implementation details. |
|
||||||
|
| Configuration contract | `docs/config.md` | Discovery and precedence, file schemas, fields, defaults, environment overrides, validation rules, and user-selectable stage or integration settings. | Complete example files, CLI syntax, runtime state lifecycle, implementation details. |
|
||||||
|
| Operations | `docs/operations.md` | Runtime workflows, physical filesystem and remote-state layout, output and diagnostic handling, resume, cleanup, permissions, recovery, and operational limits. | CLI flag syntax, configuration field definitions, logical artifact schemas, implementation mechanics. |
|
||||||
|
| Troubleshooting | `docs/troubleshooting.md` | Symptom-driven diagnosis, likely causes, safe inspection steps and remedies, and links to relevant contracts. | CLI syntax, configuration definitions, operational procedures, integration contracts, implementation mechanics. |
|
||||||
|
| Public HTTP contract, if introduced | `docs/api.md` | Routes, authentication, media types, request and response schemas, status codes, pagination, caching, idempotency, rate limits, and HTTP retry semantics. | Client walkthroughs, upstream or downstream integration internals, implementation detail. |
|
||||||
|
| Consumer guidance, if a public package or API is introduced | `docs/consumers/` | Task-oriented use of the public interface, minimal client examples, and consumer responsibilities. | HTTP wire semantics, external protocol contracts, internal implementation detail. |
|
||||||
|
| External and durable integration contracts | `docs/integrations/` | External file formats and protocols, upstream and downstream contracts, logical artifact paths and schemas, media types, and compatibility behavior. | Physical runtime placement and lifecycle, internal transformations, CLI syntax, configuration defaults. |
|
||||||
|
| Implemented component inventory | `docs/internal/overview.md` | Current packages and components, their implemented responsibilities, and links to focused internal docs. | Normative architecture, contributor reading policy, external contracts. |
|
||||||
|
| Internal component behavior | Other files under `docs/internal/` | Implementation flow, internal collaborators and state transitions, package-local guarantees and failures, and relevant tests. | Global architecture invariants, configuration definitions and defaults, external schemas, operator procedures. |
|
||||||
|
| Architectural decision history | `docs/adr/` | Significant decisions, context, alternatives, rationale, consequences, and supersession history. | Current behavior reference, implementation status, task sequencing. |
|
||||||
|
| Future work and implementation status | `docs/roadmap/` | Proposed, accepted, deferred, or rejected work; implementation status; sequencing; and task breakdowns. | Implemented behavior reference and architectural decision rationale. |
|
||||||
|
| Complete copyable artifacts | `examples/` | Maintained configuration, inputs, and other files intended to be copied or run. | Field-by-field reference, command reference, prose explanation. |
|
||||||
|
|
||||||
|
Documents that do not exist are required only when the corresponding interface
|
||||||
|
or responsibility exists. Do not create placeholder API, consumer, integration,
|
||||||
|
or operations documents for behavior the application does not have.
|
||||||
|
|
||||||
|
## Boundary Rules
|
||||||
|
|
||||||
|
### Orientation
|
||||||
|
|
||||||
|
The README owns product orientation. The developer guide routes contributors.
|
||||||
|
Architecture owns normative structure. Internal overview owns the current
|
||||||
|
concrete component map. These documents may link to one another but should not
|
||||||
|
maintain parallel package or behavior descriptions.
|
||||||
|
|
||||||
|
### Commands, Configuration, Operations, And Troubleshooting
|
||||||
|
|
||||||
|
CLI documentation answers how to invoke the application. Configuration
|
||||||
|
documentation answers what settings mean. Operations answers what happens to
|
||||||
|
runtime state and how to operate or recover the application. Troubleshooting
|
||||||
|
starts from observable symptoms and links readers to the owning command,
|
||||||
|
configuration, operational, or integration contract. When a workflow crosses
|
||||||
|
these topics, choose the document that owns the task and link to the other
|
||||||
|
contracts.
|
||||||
|
|
||||||
|
### Contracts And Implementation
|
||||||
|
|
||||||
|
Integration and API documents define externally observable shapes and
|
||||||
|
semantics. Internal documents explain how Narratio implements or consumes those
|
||||||
|
contracts. Internal docs may name a field, file, or protocol to identify a
|
||||||
|
dependency, but must link to its canonical contract for the definition.
|
||||||
|
|
||||||
|
### Security Topics
|
||||||
|
|
||||||
|
This policy owns what documentation and examples may contain. Architecture owns
|
||||||
|
application security invariants. Configuration owns credential-supply
|
||||||
|
mechanisms. Operations owns permissions and handling of sensitive runtime
|
||||||
|
artifacts. Troubleshooting owns safe diagnostic and remediation guidance.
|
||||||
|
Internal docs own implementation mechanisms only.
|
||||||
|
|
||||||
|
## Architecture Decision Records
|
||||||
|
|
||||||
|
Use sequentially numbered ADR filenames such as
|
||||||
|
`0001-record-architecture-decisions.md`. Follow the lightweight Nygard format:
|
||||||
|
|
||||||
|
1. title;
|
||||||
|
2. status;
|
||||||
|
3. date;
|
||||||
|
4. context;
|
||||||
|
5. decision;
|
||||||
|
6. alternatives considered;
|
||||||
|
7. consequences.
|
||||||
|
|
||||||
|
Treat the decision content of an accepted ADR as immutable. When a decision
|
||||||
|
changes, create a new ADR and update the earlier ADR's status to superseded.
|
||||||
|
Rejected architectural alternatives belong in the ADR; rejected product ideas
|
||||||
|
belong in the roadmap.
|
||||||
|
|
||||||
|
## Maintenance
|
||||||
|
|
||||||
|
When behavior changes, update its canonical owner in the same change. If
|
||||||
|
ownership moves, remove the old definition and replace it with a link where
|
||||||
|
navigation remains useful.
|
||||||
|
|
||||||
|
Before completing documentation work:
|
||||||
|
|
||||||
|
- verify affected behavior and examples;
|
||||||
|
- check commands, flags, fields, defaults, schemas, and paths against their
|
||||||
|
implementation;
|
||||||
|
- keep unimplemented behavior in the roadmap, subject to the ADR exception;
|
||||||
|
- remove stale references and validate links;
|
||||||
|
- confirm that non-owning documents summarize and link rather than redefine;
|
||||||
|
- confirm that no secrets or sensitive private data were added.
|
||||||
296
docs/policy/testing.md
Normal file
296
docs/policy/testing.md
Normal file
@@ -0,0 +1,296 @@
|
|||||||
|
# Testing Policy
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
|
||||||
|
Our tests exist to make **incorrect changes expensive and correct changes cheap**.
|
||||||
|
|
||||||
|
We do not optimize for test count, line coverage, exhaustive isolation, or the fewest possible tests. We optimize for sufficient confidence in important behavior while imposing as little unnecessary friction as possible on future development.
|
||||||
|
|
||||||
|
## Every test has a cost
|
||||||
|
|
||||||
|
Testing is not an unqualified good. Every test imposes both an immediate cost and a continuing lifetime cost.
|
||||||
|
|
||||||
|
A test must be:
|
||||||
|
|
||||||
|
- written and reviewed;
|
||||||
|
- understood by future maintainers and coding agents;
|
||||||
|
- executed in local and CI workflows;
|
||||||
|
- diagnosed when it fails;
|
||||||
|
- updated when legitimate behavior changes;
|
||||||
|
- maintained as fixtures, APIs, and dependencies evolve; and
|
||||||
|
- removed or rewritten when it becomes redundant, brittle, misleading, or obsolete.
|
||||||
|
|
||||||
|
Tests also create cognitive and architectural friction. They can constrain refactoring, duplicate policy, slow feedback loops, add noise to failures, and cause harmless implementation changes to require unrelated edits across the suite.
|
||||||
|
|
||||||
|
A test is warranted only when the confidence it provides justifies these costs.
|
||||||
|
|
||||||
|
Apply this cost-benefit analysis at two levels:
|
||||||
|
|
||||||
|
1. **Per test:** What realistic defect does this test detect, how consequential would that defect be, and is that protection worth the test's lifetime cost?
|
||||||
|
2. **Across the suite:** Does this collection provide materially more confidence than a smaller, simpler suite would?
|
||||||
|
|
||||||
|
The preferred test suite is a **lean suite that provides sufficient confidence in the risks that matter, without redundant or low-value tests**. We seek sufficient confidence with the least unnecessary testing friction, not the fewest possible tests.
|
||||||
|
|
||||||
|
Some friction is intentional. Tests should make dangerous changes—such as breaking compatibility, corrupting data, violating security boundaries, or reintroducing subtle bugs—require deliberate review. They should not make ordinary internal changes needlessly expensive.
|
||||||
|
|
||||||
|
The cost of a test is not a reason to omit testing by default. Do not cite maintenance cost abstractly. When omitting a plausible test, be able to state why the protected failure is low-risk, already covered, obvious, reversible, or cheaper to detect elsewhere. For consequential, subtle, or difficult-to-observe behavior, the presumption should favor testing.
|
||||||
|
|
||||||
|
## Default testing style
|
||||||
|
|
||||||
|
Use a **classical/Detroit-style** approach:
|
||||||
|
|
||||||
|
- Test observable behavior, resulting state, contracts, and invariants.
|
||||||
|
- Use real internal collaborators when they are fast and deterministic.
|
||||||
|
- Use fakes, stubs, or mocks primarily at expensive, nondeterministic, destructive, or external boundaries.
|
||||||
|
- Prefer package-level behavioral tests over tests coupled to private helpers or internal call sequences.
|
||||||
|
- Treat exact collaborator interactions as testable behavior only when the interaction itself is a requirement.
|
||||||
|
|
||||||
|
Examples of appropriate seams include clocks, randomness, subprocesses, remote APIs, object storage, email, and paid LLM calls.
|
||||||
|
|
||||||
|
## Test execution requirements
|
||||||
|
|
||||||
|
Tests in the default suite must be deterministic, offline, and independent of real credentials. They must not invoke paid APIs or depend on mutable external services. Tests that require live infrastructure must be explicitly opt-in and clearly separated from the default suite.
|
||||||
|
|
||||||
|
Control clocks, randomness, environment variables, and other process-global or machine-specific state when they affect behavior. Tests should be safe to run repeatedly and alongside other tests without depending on execution order or state left by an earlier test.
|
||||||
|
|
||||||
|
## What deserves tests
|
||||||
|
|
||||||
|
Prioritize tests for:
|
||||||
|
|
||||||
|
1. Public and package-level contracts.
|
||||||
|
2. Domain rules and important invariants.
|
||||||
|
3. Boundary conditions and malformed input.
|
||||||
|
4. Failure handling, cancellation, retries, recovery, and partial success.
|
||||||
|
5. Serialization, schemas, compatibility, and round trips.
|
||||||
|
6. Previously observed or plausible regressions.
|
||||||
|
7. Representative integration and end-to-end workflows.
|
||||||
|
|
||||||
|
A package-level contract is behavior relied upon by another package or major collaborator, not every observable detail of a package implementation.
|
||||||
|
|
||||||
|
For behavior involving **data integrity, destructive operations, compatibility, security, concurrency, idempotency, or recovery**, presume that durable tests are required unless the behavior is already credibly protected at another layer.
|
||||||
|
|
||||||
|
Do not add tests merely because a function, branch, or line exists. Do not add a test when the same meaningful risk is already adequately protected elsewhere.
|
||||||
|
|
||||||
|
## Choose the right test boundary
|
||||||
|
|
||||||
|
Test through the narrowest stable boundary that expresses the behavior clearly.
|
||||||
|
|
||||||
|
This is often the package API, but it may instead be:
|
||||||
|
|
||||||
|
- a smaller pure function when dense domain logic is most clearly isolated there;
|
||||||
|
- a package-level operation when several internal collaborators jointly produce the behavior; or
|
||||||
|
- a larger integration boundary when correctness emerges from interaction with a real dependency.
|
||||||
|
|
||||||
|
Do not force all behavior through oversized end-to-end tests. Do not test every private helper merely because it exists. Choose the boundary that gives durable confidence with the least incidental coupling.
|
||||||
|
|
||||||
|
## Test behavior, not implementation
|
||||||
|
|
||||||
|
A test should protect a decision, contract, or invariant—not memorialize the current implementation.
|
||||||
|
|
||||||
|
Before adding or retaining a test, ask:
|
||||||
|
|
||||||
|
> What realistic defect would this test catch?
|
||||||
|
|
||||||
|
A test is suspect when its main purpose is to detect that someone:
|
||||||
|
|
||||||
|
- changed an internal constant;
|
||||||
|
- renamed or split a private helper;
|
||||||
|
- reordered equivalent internal operations;
|
||||||
|
- changed incidental formatting;
|
||||||
|
- replaced one correct algorithm with another; or
|
||||||
|
- refactored internal object structure without changing behavior.
|
||||||
|
|
||||||
|
Refactoring should normally require no test edits unless the refactored structure is itself part of the contract.
|
||||||
|
|
||||||
|
A test can be factually correct and still have negative value. Accurately describing current behavior is not enough; the protected behavior must be important enough to justify the future friction.
|
||||||
|
|
||||||
|
## Expected effects of different changes
|
||||||
|
|
||||||
|
Use the following expectations when evaluating test failures and test maintenance:
|
||||||
|
|
||||||
|
| Change | Expected effect on tests |
|
||||||
|
|---|---|
|
||||||
|
| Internal refactor that preserves behavior | Existing tests should normally remain unchanged and continue to pass. |
|
||||||
|
| Change to an internal default with no contractual significance | Behavioral tests should normally remain unchanged; tests should derive expectations from configuration or relationships rather than duplicate the old value. |
|
||||||
|
| Intentional change to public behavior, policy, schema, or compatibility guarantees | The relevant tests should be reviewed and changed deliberately. |
|
||||||
|
| Accidental violation of a contract or invariant | Tests should fail; fix the production code rather than rewriting the tests to accept the defect. |
|
||||||
|
|
||||||
|
A test failing is not the same as a test needing to be edited. Many tests may correctly fail because of one production defect. The maintenance smell is a correct internal change that requires unrelated expectation updates throughout the suite.
|
||||||
|
|
||||||
|
## Separate mechanism from policy
|
||||||
|
|
||||||
|
Configurable thresholds and defaults must not be duplicated throughout the test suite.
|
||||||
|
|
||||||
|
For example, do not encode an internal concurrency limit indirectly:
|
||||||
|
|
||||||
|
```go
|
||||||
|
// Production policy:
|
||||||
|
const maxConcurrency = 4
|
||||||
|
|
||||||
|
// Brittle test:
|
||||||
|
err := startProcesses(5)
|
||||||
|
require.Error(t, err)
|
||||||
|
```
|
||||||
|
|
||||||
|
Instead, test the mechanism relationally:
|
||||||
|
|
||||||
|
```go
|
||||||
|
const limit = 2
|
||||||
|
runner := NewRunner(limit)
|
||||||
|
|
||||||
|
require.NoError(t, runner.Start(limit))
|
||||||
|
require.ErrorIs(t, runner.Start(limit+1), ErrTooMuchConcurrency)
|
||||||
|
```
|
||||||
|
|
||||||
|
The test should prove:
|
||||||
|
|
||||||
|
- the configured limit is accepted; and
|
||||||
|
- one beyond the configured limit is rejected.
|
||||||
|
|
||||||
|
The production default should be tested exactly only when its literal value is itself a public, operational, safety, protocol, or compatibility requirement.
|
||||||
|
|
||||||
|
Apply the same rule to limits, timeouts, capacities, retry counts, and ranges: test relationships and behavior, not duplicated literals.
|
||||||
|
|
||||||
|
For concurrency limits, test both kinds of behavior when relevant:
|
||||||
|
|
||||||
|
1. **Configuration enforcement:** invalid or excessive requested values are handled correctly.
|
||||||
|
2. **Runtime enforcement:** observed peak concurrency never exceeds the configured limit.
|
||||||
|
|
||||||
|
Use a test-controlled limit and measure the behavior relative to that limit. Do not merely assert today's default value.
|
||||||
|
|
||||||
|
## Avoid semantic duplication across layers
|
||||||
|
|
||||||
|
Each behavior should have a clear test owner.
|
||||||
|
|
||||||
|
- Parser tests own parsing cases.
|
||||||
|
- Validator tests own validation rules.
|
||||||
|
- Domain tests own transformations and invariants.
|
||||||
|
- Adapter tests own external integration behavior.
|
||||||
|
- Orchestrator tests own coordination and failure propagation.
|
||||||
|
- CLI tests own argument and configuration mapping.
|
||||||
|
- End-to-end tests prove that representative assembled workflows work.
|
||||||
|
|
||||||
|
Higher-level tests should not repeat every lower-level case. A single intentional policy change should not require unrelated edits across many test files.
|
||||||
|
|
||||||
|
Tests that are individually reasonable may still be collectively redundant. Evaluate the marginal value of each additional test in light of the protection already provided by the rest of the suite.
|
||||||
|
|
||||||
|
## Use test doubles deliberately
|
||||||
|
|
||||||
|
Choose the least elaborate test double that provides the required control or observation.
|
||||||
|
|
||||||
|
As a default:
|
||||||
|
|
||||||
|
1. Prefer real collaborators when they are fast and deterministic.
|
||||||
|
2. Use small in-memory fakes when realistic stateful behavior is helpful.
|
||||||
|
3. Use stubs when a dependency only needs to provide controlled responses.
|
||||||
|
4. Use mocks when the interaction itself is contractual.
|
||||||
|
|
||||||
|
Mocks are appropriate when the contract includes facts such as:
|
||||||
|
|
||||||
|
- a notification is sent exactly once;
|
||||||
|
- a transaction is committed only after successful writes;
|
||||||
|
- cancellation reaches a subprocess;
|
||||||
|
- an expensive API is called no more than once; or
|
||||||
|
- a security audit event is emitted.
|
||||||
|
|
||||||
|
Do not use mocks merely to isolate every object or reproduce the implementation's call graph.
|
||||||
|
|
||||||
|
## Go-specific guidance
|
||||||
|
|
||||||
|
Use:
|
||||||
|
|
||||||
|
- table-driven tests for meaningful behavioral categories and boundaries;
|
||||||
|
- `t.TempDir()` for real filesystem behavior;
|
||||||
|
- `httptest.Server` for realistic HTTP interactions;
|
||||||
|
- fuzz tests for parsers, normalization, path handling, and broad input spaces;
|
||||||
|
- golden files only when the complete output is intentionally stable;
|
||||||
|
- integration tests where correctness depends on component interaction; and
|
||||||
|
- a small number of representative end-to-end tests.
|
||||||
|
|
||||||
|
Avoid exact error-string assertions unless the wording is itself contractual. Prefer `errors.Is`, `errors.As`, typed errors, or structured error fields.
|
||||||
|
|
||||||
|
At CLI boundaries, prefer exit classifications, structured output, and the smallest stable semantic fragment needed to identify the error. Do not snapshot complete diagnostic wording unless it is contractual.
|
||||||
|
|
||||||
|
Golden-file updates must require an explicit local flag. CI must not update golden files automatically, and reviewers must inspect the semantic diff before accepting an update.
|
||||||
|
|
||||||
|
Keep tests readable and direct. Test helpers and fixture frameworks must earn their own maintenance cost; do not build elaborate test infrastructure for small or isolated needs.
|
||||||
|
|
||||||
|
## Coverage
|
||||||
|
|
||||||
|
Coverage is a diagnostic, not a target.
|
||||||
|
|
||||||
|
Use it to find untested critical branches and unexpectedly weak packages. Do not write low-value tests solely to increase a percentage, and do not infer test quality from coverage alone.
|
||||||
|
|
||||||
|
Pure domain logic will often warrant higher coverage than CLI wiring or external adapters. Uneven coverage is acceptable when it reflects risk.
|
||||||
|
|
||||||
|
Increasing coverage is valuable only when the newly covered behavior protects a meaningful risk at an acceptable cost.
|
||||||
|
|
||||||
|
## Regression tests
|
||||||
|
|
||||||
|
A bug fix should normally include a regression test that fails before the fix and passes afterward.
|
||||||
|
|
||||||
|
Retain the test when the defect could realistically recur and its consequences justify the ongoing cost. Prefer the narrowest durable test of the violated contract or invariant; do not preserve accidental implementation details from the original bug.
|
||||||
|
|
||||||
|
Not every historical bug requires a permanent test. If the underlying design has made recurrence impossible, the test has become redundant, or a stronger invariant test now subsumes it, remove or consolidate it.
|
||||||
|
|
||||||
|
## Deleting or rewriting tests
|
||||||
|
|
||||||
|
Tests are maintained code, not permanent historical artifacts.
|
||||||
|
|
||||||
|
Delete or rewrite a test when its maintenance cost exceeds the confidence it provides.
|
||||||
|
|
||||||
|
Strong candidates include tests that:
|
||||||
|
|
||||||
|
- require updates after harmless internal changes;
|
||||||
|
- directly assert private constants without protecting a real contract;
|
||||||
|
- duplicate the same policy across several layers;
|
||||||
|
- verify mock choreography rather than outcomes;
|
||||||
|
- snapshot large amounts of incidental output;
|
||||||
|
- test trivial private helpers already exercised through stable package behavior;
|
||||||
|
- protect risks already covered more effectively elsewhere;
|
||||||
|
- are flaky, misleading, obsolete, or disproportionately expensive to diagnose; or
|
||||||
|
- no longer correspond to a plausible failure mode.
|
||||||
|
|
||||||
|
Several brittle tests may encode one genuine requirement. Replace them with one durable behavior-level or invariant test rather than preserving all of them.
|
||||||
|
|
||||||
|
Deleting a low-value test can improve the quality of the suite by reducing noise, maintenance burden, and friction around legitimate change.
|
||||||
|
|
||||||
|
## Reviewing a proposed test
|
||||||
|
|
||||||
|
Use the following questions when the value, boundary, or durability of a proposed test is not self-evident. Significant test additions should be reviewable against them, but written answers are not required for every routine test.
|
||||||
|
|
||||||
|
1. What realistic defect would it catch?
|
||||||
|
2. How likely is that defect?
|
||||||
|
3. How consequential would it be?
|
||||||
|
4. Is the behavior already protected elsewhere?
|
||||||
|
5. At which layer should this behavior be owned?
|
||||||
|
6. Does the test assert a durable contract or an incidental implementation detail?
|
||||||
|
7. Could the implementation be refactored without changing the behavior and without editing this test?
|
||||||
|
8. What should cause this test to fail?
|
||||||
|
9. What legitimate changes should not cause this test to fail?
|
||||||
|
10. What ongoing maintenance, execution, and diagnostic cost will the test impose?
|
||||||
|
11. Is there a smaller or more direct test that protects the same risk?
|
||||||
|
|
||||||
|
Do not add the test when its expected lifetime cost exceeds its expected protective value.
|
||||||
|
|
||||||
|
When deciding not to test plausible behavior, record or be able to explain why the risk is low, already protected, obvious, reversible, or cheaper to detect elsewhere.
|
||||||
|
|
||||||
|
## Definition of sufficient
|
||||||
|
|
||||||
|
A test suite is sufficient when:
|
||||||
|
|
||||||
|
- important contracts and invariants are protected;
|
||||||
|
- meaningful boundaries and failure modes are exercised;
|
||||||
|
- realistic and consequential regressions are credibly protected against silent recurrence;
|
||||||
|
- behavior involving data integrity, destructive operations, compatibility, security, concurrency, idempotency, and recovery is credibly protected;
|
||||||
|
- important external boundaries have realistic integration coverage;
|
||||||
|
- representative complete workflows are tested;
|
||||||
|
- failures provide useful signal rather than redundant noise;
|
||||||
|
- legitimate internal changes usually do not require test edits; and
|
||||||
|
- additional tests would mostly repeat existing protection or preserve inconsequential implementation details.
|
||||||
|
|
||||||
|
Sufficiency is a risk judgment, not a coverage percentage or test count. Reassess it as the application, its users, and the consequences of failure evolve.
|
||||||
|
|
||||||
|
The governing rule is:
|
||||||
|
|
||||||
|
> Test heavily where failure is consequential, subtle, or difficult to detect after the fact. Test lightly where failure is obvious, reversible, and inexpensive—and retain no test whose lifetime cost exceeds the confidence it provides.
|
||||||
97
docs/release.md
Normal file
97
docs/release.md
Normal file
@@ -0,0 +1,97 @@
|
|||||||
|
# Releasing Narratio
|
||||||
|
|
||||||
|
This document is the maintainer procedure for creating a Narratio source and
|
||||||
|
binary release. The synchronous release boundary is a successful push of one
|
||||||
|
new tag to `origin`; Woodpecker and Gitea publication happen later and do not
|
||||||
|
change that result.
|
||||||
|
|
||||||
|
## Choose a version and write its note
|
||||||
|
|
||||||
|
Narratio is past `v1.0.0`. Use an unused stable tag in the exact form
|
||||||
|
`vMAJOR.MINOR.PATCH`:
|
||||||
|
|
||||||
|
- increment `MINOR` for backward-compatible features;
|
||||||
|
- increment `PATCH` for backward-compatible fixes; and
|
||||||
|
- reserve a new `MAJOR` for an intentional breaking documented contract.
|
||||||
|
|
||||||
|
Before preparing the candidate, create
|
||||||
|
`docs/releases/vMAJOR.MINOR.PATCH.md` with this structure:
|
||||||
|
|
||||||
|
```markdown
|
||||||
|
# Narratio vMAJOR.MINOR.PATCH
|
||||||
|
|
||||||
|
This release ...
|
||||||
|
|
||||||
|
## Summary
|
||||||
|
|
||||||
|
## Compatibility
|
||||||
|
|
||||||
|
## Upgrade
|
||||||
|
|
||||||
|
## Changes
|
||||||
|
```
|
||||||
|
|
||||||
|
The compatibility section identifies relevant CLI, configuration, artifact,
|
||||||
|
integration, or operating-contract changes. The upgrade section states the
|
||||||
|
required operator action, or explicitly says that no special action is
|
||||||
|
required. Release notes are immutable historical summaries; link to the
|
||||||
|
current canonical documentation for detailed behavior.
|
||||||
|
|
||||||
|
Commit the note and all candidate changes, then use the ordinary development
|
||||||
|
workflow to push that commit to `main`. Do not create a release tag before the
|
||||||
|
candidate is committed and `origin/main` contains the exact same commit.
|
||||||
|
|
||||||
|
## Validate the candidate
|
||||||
|
|
||||||
|
Run the shared checker from any directory:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
scripts/check-release-candidate.sh vMAJOR.MINOR.PATCH
|
||||||
|
```
|
||||||
|
|
||||||
|
It validates the version and matching note, module hygiene, formatting,
|
||||||
|
whitespace, uncached tests, race tests, static checks, documentation, examples,
|
||||||
|
and six official cross-build assets. It uses `GOWORK=off`, does not contact
|
||||||
|
application services, CI, or Gitea, and does not create tags or modify tracked
|
||||||
|
source. Fix any failure on `main`, commit it, push it normally, and rerun the
|
||||||
|
checker.
|
||||||
|
|
||||||
|
The checker builds Linux, macOS, and Windows assets for `amd64` and `arm64`.
|
||||||
|
Cross-builds prove compilation; they are not native macOS or Windows runtime
|
||||||
|
evidence.
|
||||||
|
|
||||||
|
## Publish the tag
|
||||||
|
|
||||||
|
From a clean checkout on `main` whose `HEAD` equals `origin/main`, run:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
scripts/release.sh vMAJOR.MINOR.PATCH
|
||||||
|
```
|
||||||
|
|
||||||
|
The command fetches and checks `origin/main`, re-runs candidate validation,
|
||||||
|
then fetches and checks again before creating an explicitly unsigned lightweight
|
||||||
|
tag for the originally recorded commit. It refuses dirty, divergent, changed,
|
||||||
|
or already-tagged candidates. It pushes only:
|
||||||
|
|
||||||
|
```text
|
||||||
|
refs/tags/vMAJOR.MINOR.PATCH:refs/tags/vMAJOR.MINOR.PATCH
|
||||||
|
```
|
||||||
|
|
||||||
|
It never commits changes, pushes `main`, force-pushes, moves a tag, or pushes
|
||||||
|
all tags. A successful push of that exact ref completes the release command;
|
||||||
|
the command prints the tag and commit, then returns without waiting for CI,
|
||||||
|
querying Gitea, downloading assets, or checking checksums.
|
||||||
|
|
||||||
|
If a failure occurs before the tag is created, correct the candidate on `main`
|
||||||
|
and repeat validation. If the push fails after local tag creation, the local tag
|
||||||
|
is intentionally retained for inspection and the command must not be retried
|
||||||
|
blindly. Once the upstream tag has been pushed, it is immutable. Correct any
|
||||||
|
defect or failed asynchronous publication with a new patch version, a new
|
||||||
|
release note, and the complete procedure again.
|
||||||
|
|
||||||
|
## Optional asynchronous inspection
|
||||||
|
|
||||||
|
After a successful tag push, a human may later inspect the tag-triggered
|
||||||
|
Woodpecker run and the corresponding Gitea release for binaries and checksums.
|
||||||
|
This is optional follow-up only. Automated releasers must not wait for, poll,
|
||||||
|
or treat CI/Gitea completion as a condition of the successful tag push.
|
||||||
26
docs/releases/README.md
Normal file
26
docs/releases/README.md
Normal file
@@ -0,0 +1,26 @@
|
|||||||
|
# Release Notes
|
||||||
|
|
||||||
|
This directory contains immutable historical release notes for Narratio.
|
||||||
|
Future notes are created with the matching stable version and use this minimum
|
||||||
|
structure:
|
||||||
|
|
||||||
|
```markdown
|
||||||
|
# Narratio vMAJOR.MINOR.PATCH
|
||||||
|
|
||||||
|
This release ...
|
||||||
|
|
||||||
|
## Summary
|
||||||
|
|
||||||
|
## Compatibility
|
||||||
|
|
||||||
|
## Upgrade
|
||||||
|
|
||||||
|
## Changes
|
||||||
|
```
|
||||||
|
|
||||||
|
See the [release procedure](../release.md) for creating a candidate and tag.
|
||||||
|
When asynchronous publication succeeds, the corresponding Gitea release is the
|
||||||
|
canonical source for downloadable binaries and checksums.
|
||||||
|
|
||||||
|
- [v1.6.0](v1.6.0.md)
|
||||||
|
- [v1.5.0](v1.5.0.md)
|
||||||
47
docs/releases/v1.5.0.md
Normal file
47
docs/releases/v1.5.0.md
Normal file
@@ -0,0 +1,47 @@
|
|||||||
|
# Narratio v1.5.0
|
||||||
|
|
||||||
|
Narratio v1.5.0 makes repeated post-transcript artifact development faster and
|
||||||
|
more explicit while retaining the fixed, stage-driven pipeline model.
|
||||||
|
|
||||||
|
## Highlights
|
||||||
|
|
||||||
|
- The canonical pipeline now completes deterministic rendering before
|
||||||
|
extraction, cleanly separating transcript-generating stages from
|
||||||
|
artifact-generating stages.
|
||||||
|
- `narratio run` and `narratio session plan` accept inclusive `--from` and
|
||||||
|
`--through` bounds. Excluded transcript stages are not executed or
|
||||||
|
invalidated by a bounded artifact-regeneration run.
|
||||||
|
- `narratio regenerate-artifacts SESSION` is an exact convenience alias for a
|
||||||
|
forced run from `extract` through `analyze`, including focused
|
||||||
|
`--artifacts` selections.
|
||||||
|
- Configured Scriptorium artifacts now have independent,
|
||||||
|
manifest-authoritative freshness. Narratio reuses validated current work,
|
||||||
|
rebuilds stale prerequisites in dependency order, and persists successful,
|
||||||
|
failed, and newly stale artifact state when an analysis invocation only
|
||||||
|
partially succeeds.
|
||||||
|
- Publish consumes only configured artifacts backed by current manifest
|
||||||
|
evidence; incidental or tampered files are not promoted as current output.
|
||||||
|
|
||||||
|
## Reliability And Administration
|
||||||
|
|
||||||
|
- Bounded prerequisites are checked again under the session lock before any
|
||||||
|
run mutation, closing a concurrent-run race.
|
||||||
|
- Analysis fingerprints are stable across executable and configuration path
|
||||||
|
changes and continue to cover only Narratio-observable semantic inputs.
|
||||||
|
- Runner composition now carries one validated execution plan from command
|
||||||
|
parsing through prerequisite validation, adapter composition, manifest
|
||||||
|
recording, and stage execution.
|
||||||
|
- `narratio version` reports the exact tag embedded in official release
|
||||||
|
binaries; ordinary source builds report `dev`.
|
||||||
|
|
||||||
|
## Upgrade Notes
|
||||||
|
|
||||||
|
- Existing unbounded commands and direct `run-stage`, `analyze`, and `publish`
|
||||||
|
workflows retain their meanings.
|
||||||
|
- Manifests written before artifact-level analysis state remain readable.
|
||||||
|
Legacy aggregate analysis success is not sufficient freshness evidence, so
|
||||||
|
the first analysis evaluation after upgrading may regenerate configured
|
||||||
|
artifacts once.
|
||||||
|
- Narratio cannot observe executable contents or configuration, prompt,
|
||||||
|
profile, module, and other files loaded privately by Scriptorium. Explicitly
|
||||||
|
force affected artifacts after changing those private inputs.
|
||||||
74
docs/releases/v1.6.0.md
Normal file
74
docs/releases/v1.6.0.md
Normal file
@@ -0,0 +1,74 @@
|
|||||||
|
# Narratio v1.6.0
|
||||||
|
|
||||||
|
Narratio v1.6.0 makes large pipeline configurations easier to organize,
|
||||||
|
inspect, and vary while adding character-oriented artifact generation and a
|
||||||
|
guarded, reproducible release procedure.
|
||||||
|
|
||||||
|
## Summary
|
||||||
|
|
||||||
|
Pipeline configuration can now be assembled from explicit additive imports and
|
||||||
|
a selected production or testing profile. Campaigns can own a canonical,
|
||||||
|
versioned party roster, and Scriptorium artifact families can expand one
|
||||||
|
definition into concrete per-character artifacts, dependencies, variables, and
|
||||||
|
publish rules.
|
||||||
|
|
||||||
|
New read-only configuration commands expose the fully resolved pipeline,
|
||||||
|
source provenance, semantic digest, and profile differences before a session is
|
||||||
|
run. Stage reuse now records configuration-sensitive semantic evidence so
|
||||||
|
profile or configuration changes cannot silently reuse incompatible work.
|
||||||
|
|
||||||
|
## Compatibility
|
||||||
|
|
||||||
|
This is a backward-compatible feature release. Existing monolithic pipeline
|
||||||
|
files, concrete Scriptorium artifacts, publish rules, and configurations
|
||||||
|
without profiles remain supported. When profiles are declared and no explicit
|
||||||
|
profile is selected, Narratio uses the configured production default.
|
||||||
|
|
||||||
|
An unversioned party file plus a separate players file remains available as an
|
||||||
|
isolated legacy compatibility path, but it cannot drive artifact families. New
|
||||||
|
campaigns and new party-oriented features should use the `narratio.party.v1`
|
||||||
|
schema. Existing manifests remain readable; missing legacy semantic evidence
|
||||||
|
is treated as stale rather than trusted.
|
||||||
|
|
||||||
|
Narratio continues to consume the documented Notarius D&D pipeline contract.
|
||||||
|
The release workflow still cross-compiles Linux, macOS, and Windows binaries
|
||||||
|
for `amd64` and `arm64`; cross-compilation is not native runtime evidence for
|
||||||
|
macOS or Windows.
|
||||||
|
|
||||||
|
## Upgrade
|
||||||
|
|
||||||
|
No special action is required for existing monolithic configurations that do
|
||||||
|
not adopt profiles or artifact families. On the first run after upgrading,
|
||||||
|
stages recorded by older manifests may regenerate once because those records do
|
||||||
|
not contain the new semantic configuration evidence.
|
||||||
|
|
||||||
|
To adopt the new configuration model, use the maintained
|
||||||
|
`examples/production-testing` bundle as a migration reference: split stable
|
||||||
|
settings into explicit imports, define a production default and optional
|
||||||
|
testing profile, convert campaign party data to `narratio.party.v1`, remove the
|
||||||
|
separate players file, and then introduce character artifact families. Review
|
||||||
|
the result with `narratio config validate`, `config show`, `config sources`, and
|
||||||
|
`config diff` before running a session.
|
||||||
|
|
||||||
|
Campaign, session, previous-session, and run identifiers must satisfy the
|
||||||
|
documented portable identity grammar. Existing manifests or remote state with
|
||||||
|
unsafe legacy identifiers must be migrated before use.
|
||||||
|
|
||||||
|
## Changes
|
||||||
|
|
||||||
|
- Added root-owned, non-recursive additive pipeline imports with strict,
|
||||||
|
source-aware conflict detection.
|
||||||
|
- Added named pipeline profiles with an explicit production default and
|
||||||
|
deliberate command-line selection.
|
||||||
|
- Added `config validate`, `config show`, `config sources`, and `config diff`
|
||||||
|
for read-only inspection of resolved configuration and provenance.
|
||||||
|
- Added the strict `narratio.party.v1` campaign roster, including stable
|
||||||
|
character IDs, player and character names, optional aliases, and classes.
|
||||||
|
- Added deterministic players derivation and canonical party delivery to
|
||||||
|
downstream integrations.
|
||||||
|
- Added character-oriented artifact families, corresponding member
|
||||||
|
dependencies, family selection, and generated publish policies.
|
||||||
|
- Added semantic configuration fingerprints and stage-specific resume checks
|
||||||
|
across transcript and artifact stages.
|
||||||
|
- Added shared release candidate, asset build, and guarded tag-publication
|
||||||
|
scripts, with tag-triggered publication remaining asynchronous.
|
||||||
@@ -1,762 +0,0 @@
|
|||||||
# Roadmap: Runtime-Defined Scriptorium Artifacts
|
|
||||||
|
|
||||||
## Status
|
|
||||||
|
|
||||||
Implementation roadmap for a pre-release hard cutover.
|
|
||||||
|
|
||||||
## Purpose
|
|
||||||
|
|
||||||
Narratio currently treats artifact generation as a narrow `analyze` stage that supports a hard-coded `session_recap` artifact. This roadmap describes how to generalize artifact generation so operators can define Scriptorium-backed output artifacts at runtime through `pipeline.yml`.
|
|
||||||
|
|
||||||
The goal is to keep Narratio as a fixed pipeline orchestrator while making the artifact generation step configurable, composable, deterministic, and easy to regenerate selectively.
|
|
||||||
|
|
||||||
## Desired Outcome
|
|
||||||
|
|
||||||
Operators should be able to define artifacts such as session recaps, player handouts, NPC summaries, quest logs, entity maps, or other campaign-specific outputs without changing Narratio code.
|
|
||||||
|
|
||||||
A configured artifact is declared under:
|
|
||||||
|
|
||||||
```text
|
|
||||||
pipeline.scriptorium.artifacts.<name>
|
|
||||||
```
|
|
||||||
|
|
||||||
Each configured artifact becomes a canonical runtime artifact source ID:
|
|
||||||
|
|
||||||
```text
|
|
||||||
narratio.artifact.<name>
|
|
||||||
```
|
|
||||||
|
|
||||||
For example:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
scriptorium:
|
|
||||||
artifacts:
|
|
||||||
session_recap:
|
|
||||||
enabled: true
|
|
||||||
prompt_id: dnd_session.session_recap
|
|
||||||
output_path: artifacts/session_recap.md
|
|
||||||
inputs:
|
|
||||||
transcript:
|
|
||||||
source: narratio.transcript.trimmed
|
|
||||||
required: true
|
|
||||||
```
|
|
||||||
|
|
||||||
This artifact is addressable by later artifacts as:
|
|
||||||
|
|
||||||
```text
|
|
||||||
narratio.artifact.session_recap
|
|
||||||
```
|
|
||||||
|
|
||||||
A dependent artifact can then consume it explicitly:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
scriptorium:
|
|
||||||
artifacts:
|
|
||||||
player_handout:
|
|
||||||
enabled: true
|
|
||||||
depends_on:
|
|
||||||
- session_recap
|
|
||||||
prompt_id: dnd_session.player_handout
|
|
||||||
output_path: artifacts/player_handout.md
|
|
||||||
inputs:
|
|
||||||
recap:
|
|
||||||
source: narratio.artifact.session_recap
|
|
||||||
required: true
|
|
||||||
transcript:
|
|
||||||
source: narratio.transcript.trimmed
|
|
||||||
required: true
|
|
||||||
```
|
|
||||||
|
|
||||||
## Resolved Design Decisions
|
|
||||||
|
|
||||||
The following decisions are settled for the initial implementation:
|
|
||||||
|
|
||||||
1. Configured artifact outputs must live under Narratio's internal artifact output directory, initially `artifacts/`.
|
|
||||||
2. The artifact output directory should be defined as an internal default in `internal/config/defaults.go`, but no public configuration knob should be exposed yet.
|
|
||||||
3. Artifact `output_path` should remain explicit in the initial implementation to avoid guessing file extensions or output formats.
|
|
||||||
4. A disabled artifact may still be referenced as an input if its declared output already exists on disk and passes basic validation.
|
|
||||||
5. A disabled artifact is not executable during the current analyze run.
|
|
||||||
6. Artifact-to-artifact references require an explicit `depends_on` entry. Narratio should fail fast if the dependency declaration is missing.
|
|
||||||
7. The manifest remains stage-oriented: `analyze` succeeds or fails as a full stage.
|
|
||||||
8. Analyze-stage metadata may record per-artifact output details for provenance and later resolution, but not for intra-stage resume semantics.
|
|
||||||
9. `--artifacts` should be added as a CLI filter for selective artifact generation.
|
|
||||||
10. `--artifacts` does not imply `--force`; it only changes which configured artifacts are treated as executable when `analyze` actually runs.
|
|
||||||
11. Because Narratio is still pre-release, the hard-coded `session_recap` behavior should be removed immediately rather than deprecated gradually.
|
|
||||||
|
|
||||||
## Scope
|
|
||||||
|
|
||||||
This roadmap covers:
|
|
||||||
|
|
||||||
- introducing a runtime artifact catalog;
|
|
||||||
- generalizing configured Scriptorium artifact execution;
|
|
||||||
- supporting `narratio.artifact.<name>` source IDs;
|
|
||||||
- adding explicit artifact dependencies;
|
|
||||||
- supporting disabled-but-resolvable artifact inputs;
|
|
||||||
- adding selective artifact execution via `--artifacts`;
|
|
||||||
- recording generated artifacts in analyze-stage metadata and/or manifest outputs;
|
|
||||||
- removing hard-coded `session_recap` behavior;
|
|
||||||
- updating tests and documentation.
|
|
||||||
|
|
||||||
## Non-Goals
|
|
||||||
|
|
||||||
This feature should not turn Narratio into a general workflow engine.
|
|
||||||
|
|
||||||
The initial implementation should not add:
|
|
||||||
|
|
||||||
- arbitrary shell-command artifacts;
|
|
||||||
- arbitrary user-defined stages;
|
|
||||||
- loops or conditional branching;
|
|
||||||
- automatic archive promotion of generated artifacts;
|
|
||||||
- semantic knowledge of particular artifact types;
|
|
||||||
- per-artifact resume semantics within a successful or failed analyze stage;
|
|
||||||
- automatic dependency inference without `depends_on`.
|
|
||||||
|
|
||||||
Narratio should continue to orchestrate a fixed pipeline. The configurable part is the set of Scriptorium artifact invocations performed during the `analyze` stage.
|
|
||||||
|
|
||||||
## Current State
|
|
||||||
|
|
||||||
Narratio already has several relevant pieces in place:
|
|
||||||
|
|
||||||
- `pipeline.scriptorium.artifacts` is modeled as a map of artifact definitions.
|
|
||||||
- The Scriptorium adapter already accepts generic run/render requests.
|
|
||||||
- The artifact resolver already understands canonical artifact source IDs.
|
|
||||||
- The `analyze` stage already resolves inputs, optionally runs render-debug, invokes Scriptorium, verifies output, and records metadata.
|
|
||||||
|
|
||||||
The main limitation is that `analyze` currently treats `session_recap` as the only executable artifact and rejects other enabled artifact definitions.
|
|
||||||
|
|
||||||
## Target Architecture
|
|
||||||
|
|
||||||
### Runtime Artifact Catalog
|
|
||||||
|
|
||||||
Introduce a per-run artifact catalog that tracks built-in artifacts and configured artifacts.
|
|
||||||
|
|
||||||
Conceptually:
|
|
||||||
|
|
||||||
```text
|
|
||||||
ArtifactCatalog
|
|
||||||
├── built-in artifacts
|
|
||||||
│ ├── narratio.transcript.merged
|
|
||||||
│ ├── narratio.transcript.polished
|
|
||||||
│ ├── narratio.transcript.full
|
|
||||||
│ ├── narratio.transcript.trimmed
|
|
||||||
│ └── narratio.bounds.session
|
|
||||||
│
|
|
||||||
└── configured artifacts
|
|
||||||
├── narratio.artifact.session_recap
|
|
||||||
├── narratio.artifact.player_handout
|
|
||||||
└── narratio.artifact.npc_summary
|
|
||||||
```
|
|
||||||
|
|
||||||
The catalog should distinguish between three states:
|
|
||||||
|
|
||||||
```text
|
|
||||||
planned valid configured or built-in artifact known to Narratio
|
|
||||||
available artifact has been produced or otherwise resolved
|
|
||||||
executable configured artifact selected for execution in this analyze run
|
|
||||||
```
|
|
||||||
|
|
||||||
Configured artifacts can be planned without being executable. This distinction is important for disabled artifacts and for `--artifacts` filtering.
|
|
||||||
|
|
||||||
### Configured Artifact Source IDs
|
|
||||||
|
|
||||||
Configured artifact keys map directly to source IDs:
|
|
||||||
|
|
||||||
```text
|
|
||||||
pipeline.scriptorium.artifacts.<name>
|
|
||||||
→ narratio.artifact.<name>
|
|
||||||
```
|
|
||||||
|
|
||||||
`session_recap` should no longer be a special built-in analyze artifact. Instead, it is just a conventional configured artifact key:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
scriptorium:
|
|
||||||
artifacts:
|
|
||||||
session_recap:
|
|
||||||
enabled: true
|
|
||||||
prompt_id: dnd_session.session_recap
|
|
||||||
output_path: artifacts/session_recap.md
|
|
||||||
```
|
|
||||||
|
|
||||||
`narratio.artifact.session_recap` remains valid only because `session_recap` is configured.
|
|
||||||
|
|
||||||
### Artifact Output Directory
|
|
||||||
|
|
||||||
Add an internal default artifact output directory, initially:
|
|
||||||
|
|
||||||
```text
|
|
||||||
artifacts
|
|
||||||
```
|
|
||||||
|
|
||||||
This default should live in `internal/config/defaults.go` or the existing equivalent defaults location.
|
|
||||||
|
|
||||||
For the initial implementation:
|
|
||||||
|
|
||||||
- expose no public config knob for the artifact output directory;
|
|
||||||
- require each configured artifact to provide an explicit `output_path`;
|
|
||||||
- validate that each configured artifact `output_path` is run-relative;
|
|
||||||
- validate that each configured artifact `output_path` is under the internal artifact output directory;
|
|
||||||
- reject output paths that escape the run workspace or use path traversal.
|
|
||||||
|
|
||||||
This preserves future configurability without forcing Narratio to guess output extensions or formats now.
|
|
||||||
|
|
||||||
### Enabled, Disabled, and Selected Artifacts
|
|
||||||
|
|
||||||
Configured artifacts should have three distinct execution states:
|
|
||||||
|
|
||||||
```text
|
|
||||||
enabled by config artifact has enabled: true
|
|
||||||
selected for execution artifact remains executable after --artifacts filtering
|
|
||||||
disabled for execution artifact is not executable, but may be resolvable from disk
|
|
||||||
```
|
|
||||||
|
|
||||||
Without `--artifacts`, all configured artifacts with `enabled: true` are selected for execution.
|
|
||||||
|
|
||||||
With `--artifacts`, only the named artifacts are selected for execution. All other configured artifacts are treated as disabled for the current analyze invocation, regardless of their configured `enabled` value.
|
|
||||||
|
|
||||||
Disabled artifacts may still be resolved as inputs if their configured `output_path` exists on disk and passes validation.
|
|
||||||
|
|
||||||
### Disabled Artifact Resolution
|
|
||||||
|
|
||||||
If artifact `B` references artifact `A`, and `A` is disabled for execution, Narratio should attempt to resolve `A` from disk.
|
|
||||||
|
|
||||||
This should succeed only when:
|
|
||||||
|
|
||||||
1. `A` is defined in `pipeline.scriptorium.artifacts`;
|
|
||||||
2. `A` has a valid `output_path`;
|
|
||||||
3. the output path exists in the current run workspace;
|
|
||||||
4. the output is non-empty, or otherwise passes any available artifact-specific validation.
|
|
||||||
|
|
||||||
The resolved provenance should make the source clear, for example:
|
|
||||||
|
|
||||||
```text
|
|
||||||
filesystem.disabled_artifact_output
|
|
||||||
```
|
|
||||||
|
|
||||||
If the file does not exist or fails validation, the dependent artifact should fail before invoking Scriptorium.
|
|
||||||
|
|
||||||
Example error wording:
|
|
||||||
|
|
||||||
```text
|
|
||||||
artifact player_handout requires narratio.artifact.session_recap, but session_recap is disabled for execution and artifacts/session_recap.md does not exist
|
|
||||||
```
|
|
||||||
|
|
||||||
### Explicit Dependencies
|
|
||||||
|
|
||||||
Artifact-to-artifact references require explicit `depends_on` entries.
|
|
||||||
|
|
||||||
If artifact `B` has an input source of `narratio.artifact.A`, then `B.depends_on` must include `A`.
|
|
||||||
|
|
||||||
This should fail:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
scriptorium:
|
|
||||||
artifacts:
|
|
||||||
player_handout:
|
|
||||||
enabled: true
|
|
||||||
prompt_id: dnd_session.player_handout
|
|
||||||
output_path: artifacts/player_handout.md
|
|
||||||
inputs:
|
|
||||||
recap:
|
|
||||||
source: narratio.artifact.session_recap
|
|
||||||
required: true
|
|
||||||
```
|
|
||||||
|
|
||||||
This should pass:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
scriptorium:
|
|
||||||
artifacts:
|
|
||||||
player_handout:
|
|
||||||
enabled: true
|
|
||||||
depends_on:
|
|
||||||
- session_recap
|
|
||||||
prompt_id: dnd_session.player_handout
|
|
||||||
output_path: artifacts/player_handout.md
|
|
||||||
inputs:
|
|
||||||
recap:
|
|
||||||
source: narratio.artifact.session_recap
|
|
||||||
required: true
|
|
||||||
```
|
|
||||||
|
|
||||||
`depends_on` values refer to configured artifact keys, not full source IDs.
|
|
||||||
|
|
||||||
Dependency validation should fail on:
|
|
||||||
|
|
||||||
- references to unknown artifact keys;
|
|
||||||
- missing `depends_on` entries for artifact-to-artifact input references;
|
|
||||||
- self-dependencies;
|
|
||||||
- dependency cycles among executable artifacts.
|
|
||||||
|
|
||||||
Dependencies on disabled artifacts are permitted, but the disabled dependency must resolve from disk before the dependent artifact runs.
|
|
||||||
|
|
||||||
### Execution Order
|
|
||||||
|
|
||||||
The analyze stage should execute selected artifacts in dependency order.
|
|
||||||
|
|
||||||
Rules:
|
|
||||||
|
|
||||||
- selected artifacts are executable;
|
|
||||||
- disabled artifacts are never executed;
|
|
||||||
- selected artifacts may depend on other selected artifacts;
|
|
||||||
- selected artifacts may depend on disabled artifacts if those disabled artifacts resolve from disk;
|
|
||||||
- independent selected artifacts run in deterministic sorted-name order.
|
|
||||||
|
|
||||||
Use topological sorting over selected artifacts, while validating dependency references across the full configured artifact set.
|
|
||||||
|
|
||||||
### Input Resolution
|
|
||||||
|
|
||||||
Input resolution should use the artifact catalog and existing artifact resolver behavior.
|
|
||||||
|
|
||||||
For each configured artifact input:
|
|
||||||
|
|
||||||
- built-in sources resolve through existing resolver behavior;
|
|
||||||
- `previous_session_artifact` preserves existing behavior;
|
|
||||||
- `narratio.artifact.<name>` resolves through the runtime artifact catalog;
|
|
||||||
- selected dependencies resolve after being produced earlier in the same analyze execution;
|
|
||||||
- disabled dependencies resolve from their configured output path on disk;
|
|
||||||
- optional missing inputs are omitted;
|
|
||||||
- required missing inputs fail before Scriptorium is invoked.
|
|
||||||
|
|
||||||
### Analyze Stage Generalization
|
|
||||||
|
|
||||||
The `analyze` stage should become the generic Scriptorium artifact stage.
|
|
||||||
|
|
||||||
High-level flow:
|
|
||||||
|
|
||||||
1. Load configured Scriptorium artifacts.
|
|
||||||
2. Apply the `--artifacts` filter, if present.
|
|
||||||
3. If no artifacts are selected for execution, return success metadata with `skipped=true`.
|
|
||||||
4. Build the runtime artifact catalog.
|
|
||||||
5. Validate artifact names, output paths, source IDs, dependencies, selected artifacts, and required fields.
|
|
||||||
6. Resolve any disabled dependencies that are required by selected artifacts.
|
|
||||||
7. Sort selected artifacts by dependency order.
|
|
||||||
8. For each selected artifact:
|
|
||||||
- resolve configured inputs;
|
|
||||||
- build the Scriptorium run request;
|
|
||||||
- optionally run Scriptorium render-debug;
|
|
||||||
- run Scriptorium;
|
|
||||||
- fail on validation-failed result;
|
|
||||||
- verify the output exists and is non-empty;
|
|
||||||
- record artifact output metadata;
|
|
||||||
- register `narratio.artifact.<name>` as available in the catalog.
|
|
||||||
9. Return aggregate analyze-stage metadata containing all generated and reused artifacts relevant to the run.
|
|
||||||
|
|
||||||
The Scriptorium adapter should remain generic. It should not decide which artifacts run, how dependencies work, or how artifacts are registered.
|
|
||||||
|
|
||||||
### Manifest and Metadata
|
|
||||||
|
|
||||||
The manifest should remain stage-oriented.
|
|
||||||
|
|
||||||
This means:
|
|
||||||
|
|
||||||
- `analyze` succeeds or fails as a full stage;
|
|
||||||
- if `analyze` has already succeeded and the user does not force it, the runner skips it as a full stage;
|
|
||||||
- Narratio should not implement per-artifact resume in the first version.
|
|
||||||
|
|
||||||
However, analyze-stage metadata should still record artifact outputs for provenance and future resolution.
|
|
||||||
|
|
||||||
Recommended metadata shape:
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"skipped": false,
|
|
||||||
"artifacts": [
|
|
||||||
{
|
|
||||||
"name": "session_recap",
|
|
||||||
"source_id": "narratio.artifact.session_recap",
|
|
||||||
"output_kind": "scriptorium_artifact",
|
|
||||||
"path": "artifacts/session_recap.md",
|
|
||||||
"prompt_id": "dnd_session.session_recap",
|
|
||||||
"profile_id": "local-gemma-31b",
|
|
||||||
"provenance": "generated.current_analyze_run"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"name": "player_handout",
|
|
||||||
"source_id": "narratio.artifact.player_handout",
|
|
||||||
"output_kind": "scriptorium_artifact",
|
|
||||||
"path": "artifacts/player_handout.md",
|
|
||||||
"prompt_id": "dnd_session.player_handout",
|
|
||||||
"profile_id": "local-gemma-31b",
|
|
||||||
"provenance": "generated.current_analyze_run"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"reused_artifacts": [
|
|
||||||
{
|
|
||||||
"name": "session_recap",
|
|
||||||
"source_id": "narratio.artifact.session_recap",
|
|
||||||
"path": "artifacts/session_recap.md",
|
|
||||||
"provenance": "filesystem.disabled_artifact_output"
|
|
||||||
}
|
|
||||||
]
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
The exact struct can differ from this example, but it should preserve:
|
|
||||||
|
|
||||||
- artifact name;
|
|
||||||
- canonical source ID;
|
|
||||||
- output path;
|
|
||||||
- prompt/profile provenance for generated artifacts;
|
|
||||||
- reused-vs-generated provenance.
|
|
||||||
|
|
||||||
### Resume and Force Behavior
|
|
||||||
|
|
||||||
Keep resume behavior stage-level.
|
|
||||||
|
|
||||||
Recommended semantics:
|
|
||||||
|
|
||||||
```text
|
|
||||||
No --force, analyze already succeeded:
|
|
||||||
runner skips analyze, regardless of --artifacts.
|
|
||||||
|
|
||||||
--force, no --artifacts:
|
|
||||||
analyze regenerates all configured artifacts with enabled: true.
|
|
||||||
|
|
||||||
--force --artifacts player_handout:
|
|
||||||
analyze treats only player_handout as executable.
|
|
||||||
all other configured artifacts are disabled for execution.
|
|
||||||
disabled dependencies may be reused from disk.
|
|
||||||
|
|
||||||
--artifacts player_handout on a not-yet-completed analyze stage:
|
|
||||||
analyze runs only player_handout.
|
|
||||||
disabled dependencies may be reused from disk.
|
|
||||||
```
|
|
||||||
|
|
||||||
`--artifacts` should not imply `--force`. It is an execution filter, not a resume override.
|
|
||||||
|
|
||||||
### `--artifacts` CLI Flag
|
|
||||||
|
|
||||||
Add an `--artifacts` flag to commands that can execute or resume the analyze stage.
|
|
||||||
|
|
||||||
The flag should accept one or more configured artifact names. Internally, normalize values to a set of artifact keys.
|
|
||||||
|
|
||||||
Recommended behavior:
|
|
||||||
|
|
||||||
- validate all requested artifact names against `pipeline.scriptorium.artifacts`;
|
|
||||||
- reject unknown artifact names before running stages;
|
|
||||||
- treat requested artifacts as the only executable artifacts for the analyze stage;
|
|
||||||
- treat all other configured artifacts as disabled for execution;
|
|
||||||
- allow disabled artifacts to satisfy dependencies from disk as described above;
|
|
||||||
- if `--artifacts` is used while executing a stage other than `analyze`, either reject it or ignore it with a clear validation error. Prefer rejection.
|
|
||||||
|
|
||||||
The exact CLI parsing style can follow Narratio's existing conventions. Both comma-separated and repeatable values are acceptable if the CLI package supports them cleanly, but the internal representation should be a set of artifact keys.
|
|
||||||
|
|
||||||
### Archive Behavior
|
|
||||||
|
|
||||||
Do not automatically archive every generated artifact.
|
|
||||||
|
|
||||||
Artifact generation and archive promotion should remain separate concerns. Operators should continue to use `archive.promote_artifacts` to decide which generated files should be promoted or uploaded.
|
|
||||||
|
|
||||||
Example:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
archive:
|
|
||||||
promote_artifacts:
|
|
||||||
- from: artifacts/session_recap.md
|
|
||||||
to: artifacts/session_recap.md
|
|
||||||
required: true
|
|
||||||
- from: artifacts/player_handout.md
|
|
||||||
to: artifacts/player_handout.md
|
|
||||||
required: false
|
|
||||||
```
|
|
||||||
|
|
||||||
A later enhancement may add opt-in automatic promotion of configured artifacts, but explicit promotion should remain the default.
|
|
||||||
|
|
||||||
## Implementation Plan
|
|
||||||
|
|
||||||
### Phase 1: Config Model and Defaults
|
|
||||||
|
|
||||||
Add or update the configured artifact model to include:
|
|
||||||
|
|
||||||
- `enabled`;
|
|
||||||
- `depends_on`;
|
|
||||||
- `prompt_id`;
|
|
||||||
- `profile_id`;
|
|
||||||
- `output_path`;
|
|
||||||
- `timeout`;
|
|
||||||
- `render_debug`;
|
|
||||||
- `inputs`;
|
|
||||||
- `vars`.
|
|
||||||
|
|
||||||
Add an internal default artifact output directory in `internal/config/defaults.go`, initially set to `artifacts`.
|
|
||||||
|
|
||||||
Validation rules:
|
|
||||||
|
|
||||||
- artifact names must match a conservative identifier pattern such as `^[a-z][a-z0-9_]*$`;
|
|
||||||
- selected/executable artifacts require `prompt_id` and `output_path`;
|
|
||||||
- configured artifacts that may be referenced while disabled require `output_path`;
|
|
||||||
- configured artifact output paths must be run-relative;
|
|
||||||
- configured artifact output paths must live under the internal artifact output directory;
|
|
||||||
- configured artifact output paths must not escape the run workspace;
|
|
||||||
- `narratio.artifact.<name>` input sources must refer to configured artifact keys;
|
|
||||||
- any `narratio.artifact.<name>` input source must have a matching `depends_on` entry;
|
|
||||||
- `depends_on` entries must refer to configured artifact keys;
|
|
||||||
- dependencies must not contain self-references or executable cycles;
|
|
||||||
- input names and var names must remain compatible with the Scriptorium adapter's validation rules;
|
|
||||||
- unknown YAML fields must continue to fail strict decode.
|
|
||||||
|
|
||||||
Tests:
|
|
||||||
|
|
||||||
- valid single configured artifact;
|
|
||||||
- valid multiple independent artifacts;
|
|
||||||
- valid artifact-to-artifact dependency;
|
|
||||||
- valid dependency on disabled artifact with output path;
|
|
||||||
- invalid artifact name;
|
|
||||||
- missing required fields;
|
|
||||||
- output path outside `artifacts/`;
|
|
||||||
- dependency on missing artifact;
|
|
||||||
- missing `depends_on` for artifact input source;
|
|
||||||
- self-dependency;
|
|
||||||
- cycle detection;
|
|
||||||
- typo in `narratio.artifact.<name>` source;
|
|
||||||
- unknown YAML fields still fail strict decode.
|
|
||||||
|
|
||||||
### Phase 2: CLI Filtering
|
|
||||||
|
|
||||||
Add the `--artifacts` flag and carry the selected artifact set into the run execution options.
|
|
||||||
|
|
||||||
Implementation notes:
|
|
||||||
|
|
||||||
- parse values according to existing CLI conventions;
|
|
||||||
- normalize to artifact key strings;
|
|
||||||
- validate against configured artifact definitions after config load;
|
|
||||||
- make the selected set available to the analyze stage;
|
|
||||||
- reject use with commands or stages where analyze cannot run.
|
|
||||||
|
|
||||||
Tests:
|
|
||||||
|
|
||||||
- no `--artifacts` means all enabled artifacts are selected;
|
|
||||||
- one requested artifact is selected;
|
|
||||||
- multiple requested artifacts are selected;
|
|
||||||
- unknown requested artifact fails;
|
|
||||||
- `--artifacts` does not imply `--force`;
|
|
||||||
- `--artifacts` with already-succeeded analyze stage is skipped unless forced;
|
|
||||||
- `--artifacts` on unsupported stage command fails clearly.
|
|
||||||
|
|
||||||
### Phase 3: Runtime Artifact Catalog
|
|
||||||
|
|
||||||
Introduce an internal artifact catalog abstraction.
|
|
||||||
|
|
||||||
Responsibilities:
|
|
||||||
|
|
||||||
- register built-in artifact definitions;
|
|
||||||
- register configured artifact definitions;
|
|
||||||
- map configured artifact keys to `narratio.artifact.<name>` IDs;
|
|
||||||
- track planned, available, and executable artifact states;
|
|
||||||
- expose lookup by canonical source ID;
|
|
||||||
- record generated provenance;
|
|
||||||
- record disabled-from-disk provenance.
|
|
||||||
|
|
||||||
Keep the catalog narrow. It should not execute Scriptorium and should not understand prompt semantics.
|
|
||||||
|
|
||||||
Tests:
|
|
||||||
|
|
||||||
- built-in source lookup;
|
|
||||||
- configured source registration;
|
|
||||||
- duplicate/conflicting source handling;
|
|
||||||
- planned but unavailable artifact lookup;
|
|
||||||
- selected artifact state;
|
|
||||||
- disabled artifact state;
|
|
||||||
- registering an artifact as available after generation;
|
|
||||||
- registering a disabled artifact as available from disk;
|
|
||||||
- resolving a configured artifact from analyze metadata if that behavior is implemented.
|
|
||||||
|
|
||||||
### Phase 4: Resolver Integration
|
|
||||||
|
|
||||||
Update artifact resolution so configured artifact IDs are resolved through the runtime catalog.
|
|
||||||
|
|
||||||
Resolution behavior:
|
|
||||||
|
|
||||||
- built-in sources continue using existing resolver behavior;
|
|
||||||
- configured artifact sources resolve from catalog availability/provenance;
|
|
||||||
- selected configured artifacts become available after generation;
|
|
||||||
- disabled configured artifacts may become available from disk;
|
|
||||||
- missing optional configured artifact inputs are omitted;
|
|
||||||
- missing required configured artifact inputs fail clearly.
|
|
||||||
|
|
||||||
Tests:
|
|
||||||
|
|
||||||
- configured artifact consumes a built-in transcript source;
|
|
||||||
- configured artifact consumes another configured artifact produced earlier in the same analyze run;
|
|
||||||
- configured artifact consumes a disabled artifact resolved from disk;
|
|
||||||
- required disabled artifact missing on disk fails;
|
|
||||||
- required configured artifact missing fails;
|
|
||||||
- optional missing configured artifact is omitted;
|
|
||||||
- reused artifact provenance is recorded distinctly from generated artifact provenance.
|
|
||||||
|
|
||||||
### Phase 5: Analyze Stage Generalization
|
|
||||||
|
|
||||||
Refactor `analyze` to execute selected configured artifacts.
|
|
||||||
|
|
||||||
Implementation notes:
|
|
||||||
|
|
||||||
- remove the hard-coded `session_recap` selection path;
|
|
||||||
- remove the hard-coded rejection of non-`session_recap` artifacts;
|
|
||||||
- preserve skip behavior when Scriptorium config is absent or no artifacts are selected;
|
|
||||||
- build the runtime artifact catalog;
|
|
||||||
- apply `--artifacts` filtering;
|
|
||||||
- validate selected artifacts and their dependencies;
|
|
||||||
- pre-resolve disabled dependencies from disk where required;
|
|
||||||
- compute deterministic dependency order;
|
|
||||||
- execute selected artifacts one at a time in dependency order;
|
|
||||||
- keep render-debug behavior at global and artifact levels;
|
|
||||||
- keep Scriptorium adapter invocation generic;
|
|
||||||
- after each successful run, register the artifact as available in the catalog;
|
|
||||||
- aggregate generated and reused artifact metadata.
|
|
||||||
|
|
||||||
Tests:
|
|
||||||
|
|
||||||
- no Scriptorium config skips;
|
|
||||||
- empty artifact map skips;
|
|
||||||
- no selected artifacts skips;
|
|
||||||
- disabled artifacts do not run;
|
|
||||||
- one selected artifact runs;
|
|
||||||
- multiple independent artifacts run in deterministic order;
|
|
||||||
- dependent selected artifact receives prior selected artifact as input;
|
|
||||||
- dependent selected artifact receives disabled-from-disk artifact as input;
|
|
||||||
- render-debug works for configured artifacts;
|
|
||||||
- Scriptorium validation failure fails the stage;
|
|
||||||
- missing required input fails the stage;
|
|
||||||
- successful outputs are non-empty and recorded;
|
|
||||||
- artifact filter executes only requested artifacts.
|
|
||||||
|
|
||||||
### Phase 6: Manifest and Stage Metadata
|
|
||||||
|
|
||||||
Update analyze-stage metadata and manifest output recording to support dynamic configured artifacts.
|
|
||||||
|
|
||||||
Recommended behavior:
|
|
||||||
|
|
||||||
- every generated configured artifact gets `source_id: narratio.artifact.<name>`;
|
|
||||||
- every generated configured artifact gets a generic output kind such as `scriptorium_artifact`;
|
|
||||||
- reused disabled artifacts are recorded separately from generated artifacts;
|
|
||||||
- metadata is sufficient for debugging, provenance, and future resolver support;
|
|
||||||
- metadata does not create per-artifact resume semantics.
|
|
||||||
|
|
||||||
Because this is a pre-release hard cutover, do not preserve a special legacy `session_recap` output kind unless a current internal test or archive path still requires it temporarily. Prefer updating tests and examples to treat `session_recap` as an ordinary configured artifact.
|
|
||||||
|
|
||||||
Tests:
|
|
||||||
|
|
||||||
- metadata records one generated configured artifact;
|
|
||||||
- metadata records multiple generated configured artifacts;
|
|
||||||
- metadata records reused disabled artifact provenance;
|
|
||||||
- `session_recap` is recorded as a normal configured artifact;
|
|
||||||
- manifest still treats `analyze` as a single succeeded or failed stage;
|
|
||||||
- runner skip behavior remains stage-level.
|
|
||||||
|
|
||||||
### Phase 7: Archive and Promotion Review
|
|
||||||
|
|
||||||
Review archive behavior after dynamic artifacts are recorded.
|
|
||||||
|
|
||||||
Implementation notes:
|
|
||||||
|
|
||||||
- do not automatically promote every configured artifact;
|
|
||||||
- keep `archive.promote_artifacts` explicit;
|
|
||||||
- update default or example promotion rules to use configured `session_recap` output path;
|
|
||||||
- ensure required promotion rules fail clearly when selected artifact generation did not produce a required file.
|
|
||||||
|
|
||||||
Tests:
|
|
||||||
|
|
||||||
- generated artifact can be promoted by explicit archive rule;
|
|
||||||
- required archive promotion fails if selected artifact was not generated and no file exists;
|
|
||||||
- optional archive promotion skips cleanly if file is absent;
|
|
||||||
- hard cutover does not rely on hard-coded `session_recap` generation.
|
|
||||||
|
|
||||||
### Phase 8: Documentation and Examples
|
|
||||||
|
|
||||||
Status: complete.
|
|
||||||
|
|
||||||
Update documentation after the implementation is complete.
|
|
||||||
|
|
||||||
Recommended documentation changes:
|
|
||||||
|
|
||||||
- update `docs/config.md` with the generalized artifact configuration model;
|
|
||||||
- update `docs/internal/artifacts.md` to describe the runtime artifact catalog;
|
|
||||||
- update `docs/stages/analyze.md` to describe generic Scriptorium artifact generation;
|
|
||||||
- update Scriptorium integration docs only if the adapter contract changes;
|
|
||||||
- update full annotated pipeline examples;
|
|
||||||
- add at least one example with multiple artifacts and one dependency;
|
|
||||||
- document `--artifacts` behavior and its relationship to `--force`;
|
|
||||||
- remove documentation stating that only `session_recap` is supported.
|
|
||||||
|
|
||||||
Documentation should make clear that:
|
|
||||||
|
|
||||||
- configured artifact source IDs use `narratio.artifact.<name>`;
|
|
||||||
- `depends_on` uses artifact keys, not full source IDs;
|
|
||||||
- artifact-to-artifact source references require explicit `depends_on`;
|
|
||||||
- disabled artifacts can be reused from disk when required by selected artifacts;
|
|
||||||
- `--artifacts` filters execution but does not imply `--force`;
|
|
||||||
- archive promotion remains explicit;
|
|
||||||
- per-artifact resume is not part of the initial implementation.
|
|
||||||
|
|
||||||
## Migration Strategy
|
|
||||||
|
|
||||||
Because Narratio is pre-release, perform a hard cutover.
|
|
||||||
|
|
||||||
Required changes:
|
|
||||||
|
|
||||||
1. Remove the hard-coded `session_recap` analyze behavior.
|
|
||||||
2. Require `session_recap` to be declared under `pipeline.scriptorium.artifacts.session_recap` if the operator wants a session recap.
|
|
||||||
3. Treat `narratio.artifact.session_recap` as valid only when `session_recap` is a configured artifact key.
|
|
||||||
4. Update config examples to show `session_recap` as a normal configured artifact.
|
|
||||||
5. Update tests to stop assuming that `session_recap` is a built-in analyze artifact.
|
|
||||||
6. Keep archive promotion explicit and path-based.
|
|
||||||
|
|
||||||
Example replacement config:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
scriptorium:
|
|
||||||
binary: scriptorium
|
|
||||||
config_path: /etc/scriptorium/config.yml
|
|
||||||
timeout: 10m
|
|
||||||
render_debug: false
|
|
||||||
artifacts:
|
|
||||||
session_recap:
|
|
||||||
enabled: true
|
|
||||||
prompt_id: dnd_session.session_recap
|
|
||||||
profile_id: local-gemma-31b
|
|
||||||
output_path: artifacts/session_recap.md
|
|
||||||
timeout: 20m
|
|
||||||
inputs:
|
|
||||||
transcript:
|
|
||||||
source: narratio.transcript.trimmed
|
|
||||||
required: true
|
|
||||||
prior_recap:
|
|
||||||
source: previous_session_artifact
|
|
||||||
artifact: artifacts/session_recap.md
|
|
||||||
required: false
|
|
||||||
vars:
|
|
||||||
artifact_title: Session Recap
|
|
||||||
```
|
|
||||||
|
|
||||||
## Acceptance Criteria
|
|
||||||
|
|
||||||
The feature is complete when:
|
|
||||||
|
|
||||||
- operators can define more than one enabled Scriptorium artifact in `pipeline.yml`;
|
|
||||||
- Narratio runs selected artifacts in deterministic dependency order;
|
|
||||||
- configured artifacts are addressable as `narratio.artifact.<name>`;
|
|
||||||
- one configured artifact can consume another configured artifact as an input;
|
|
||||||
- artifact-to-artifact input references require explicit `depends_on`;
|
|
||||||
- disabled artifacts can satisfy dependencies from existing on-disk outputs;
|
|
||||||
- missing required disabled artifacts fail clearly;
|
|
||||||
- optional missing inputs are omitted;
|
|
||||||
- `--artifacts` can selectively execute valid configured artifact names;
|
|
||||||
- `--artifacts` does not imply `--force`;
|
|
||||||
- render-debug behavior works for all configured artifacts;
|
|
||||||
- generated and reused artifacts are recorded in analyze-stage metadata;
|
|
||||||
- `session_recap` is no longer hard-coded and works as a normal configured artifact;
|
|
||||||
- archive promotion remains explicit;
|
|
||||||
- tests cover config validation, dependency sorting, disabled artifact resolution, resolver behavior, CLI filtering, analyze execution, archive interactions, and metadata.
|
|
||||||
|
|
||||||
## Suggested Implementation Order
|
|
||||||
|
|
||||||
1. Config model, defaults, and validation.
|
|
||||||
2. CLI parsing and propagation of `--artifacts` selection.
|
|
||||||
3. Runtime artifact catalog.
|
|
||||||
4. Resolver integration for configured artifacts.
|
|
||||||
5. Analyze stage generalization.
|
|
||||||
6. Stage metadata and manifest output recording.
|
|
||||||
7. Archive behavior review.
|
|
||||||
8. Documentation and examples.
|
|
||||||
|
|
||||||
This order keeps the most static pieces first, then moves into execution behavior once the configuration contract is explicit and well tested.
|
|
||||||
@@ -1,289 +1,704 @@
|
|||||||
# Troubleshooting
|
# Troubleshooting
|
||||||
|
|
||||||
## Purpose
|
Operational diagnosis guide for common Narratio failures.
|
||||||
Canonical operator troubleshooting guide for recurring implemented Narratio failures.
|
|
||||||
|
|
||||||
## Config file discovery failure
|
## Config file not found
|
||||||
|
|
||||||
Symptom:
|
Symptom:
|
||||||
- `run`, `plan`, `resume`, or `run-stage` fails with config/session not found.
|
|
||||||
|
|
||||||
Likely Cause:
|
- command fails to resolve `pipeline.yml`, `campaign.yml`, or `session.yml`.
|
||||||
- `pipeline.yml` or `session.yml` is missing from discovery paths.
|
|
||||||
- wrong working directory when relying on `./session.yml`.
|
Likely causes:
|
||||||
|
|
||||||
|
- missing files in default search paths;
|
||||||
|
- wrong campaign selection;
|
||||||
|
- omitted explicit flags.
|
||||||
|
|
||||||
Diagnostics:
|
Diagnostics:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
pwd
|
narratio session plan 2026-04-04
|
||||||
ls -l ./session.yml
|
|
||||||
ls -l /usr/local/etc/narratio/pipeline.yml /etc/narratio/pipeline.yml
|
|
||||||
```
|
```
|
||||||
|
|
||||||
Safe Fix:
|
Safe fix:
|
||||||
- pass explicit `--config` and `--session`.
|
|
||||||
- or place files in documented discovery paths.
|
|
||||||
|
|
||||||
Links:
|
- pass explicit `--config`, `--campaign` or `--campaign-file`, and `--session`.
|
||||||
- [docs/config.md](./config.md)
|
|
||||||
- [docs/cli.md](./cli.md)
|
|
||||||
|
|
||||||
## Session template rendering failure
|
Relevant reference: [Configuration discovery](./config.md#discovery-and-selection).
|
||||||
|
|
||||||
|
## Session template placeholders rejected
|
||||||
|
|
||||||
Symptom:
|
Symptom:
|
||||||
- load fails with unresolved placeholder or `session_id` mismatch.
|
|
||||||
|
|
||||||
Likely Cause:
|
- load error says session file must be concrete or contains `{{ ... }}` placeholders.
|
||||||
- templated `session.yml` used without `--session-id`.
|
|
||||||
- rendered `session_id` differs from passed `--session-id`.
|
Likely cause:
|
||||||
|
|
||||||
|
- using template content as runtime session config.
|
||||||
|
|
||||||
Diagnostics:
|
Diagnostics:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio plan --session ./session.yml --session-id 2026-04-04
|
narratio session validate 2026-04-04 --session /path/session.yml
|
||||||
```
|
```
|
||||||
|
|
||||||
Safe Fix:
|
Safe fix:
|
||||||
- pass `--session-id` when template placeholders are present.
|
|
||||||
- ensure rendered `session_id` matches intended run session id.
|
|
||||||
|
|
||||||
Links:
|
- generate concrete session YAML with `narratio session init`.
|
||||||
- [docs/config.md](./config.md)
|
|
||||||
|
|
||||||
## Strict YAML decode or validation failure
|
Relevant reference: [Operations: Session Initialization](./operations.md#session-initialization).
|
||||||
|
|
||||||
|
## Strict decode or schema validation failure
|
||||||
|
|
||||||
Symptom:
|
Symptom:
|
||||||
- config load fails with unknown field or validation error.
|
|
||||||
|
|
||||||
Likely Cause:
|
- unknown field / invalid value error during config load.
|
||||||
- typo/stale field name.
|
|
||||||
- missing required fields or invalid constraints.
|
Likely cause:
|
||||||
|
|
||||||
|
- stale field name, typo, invalid enum, or invalid duration/path format.
|
||||||
|
|
||||||
Diagnostics:
|
Diagnostics:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio plan --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04
|
narratio session plan 2026-04-04 --config /path/pipeline.yml --campaign-file /path/campaign.yml --session /path/session.yml
|
||||||
```
|
```
|
||||||
|
|
||||||
Safe Fix:
|
Safe fix:
|
||||||
- align fields/values to canonical config reference and examples.
|
|
||||||
|
|
||||||
Links:
|
- align config with [Configuration](./config.md) and the
|
||||||
- [docs/config.md](./config.md)
|
[maintained examples](../examples/README.md).
|
||||||
- [examples/](../examples/)
|
|
||||||
|
|
||||||
## `--artifacts` selection failure
|
Relevant reference: [Configuration](./config.md).
|
||||||
|
|
||||||
|
## Unexpected imported or profile value
|
||||||
|
|
||||||
Symptom:
|
Symptom:
|
||||||
- `run`/`resume`/`run-stage` fails with invalid or unknown artifact selection.
|
|
||||||
|
|
||||||
Likely Cause:
|
- an effective configuration value differs from the root file, or a duplicate
|
||||||
- `--artifacts` contains blank names or unknown artifact keys.
|
ownership/configuration error is hard to locate.
|
||||||
- `pipeline.scriptorium.artifacts` missing while using `--artifacts`.
|
|
||||||
|
|
||||||
Diagnostics:
|
Diagnostics:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio run --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 --artifacts player_handout
|
narratio config sources --config /path/pipeline.yml --profile testing
|
||||||
```
|
```
|
||||||
|
|
||||||
Safe Fix:
|
Add `--campaign` or `--campaign-file` when the pipeline has party-driven
|
||||||
- use configured artifact keys only.
|
artifact families. The output identifies each effective logical field's root,
|
||||||
- ensure `pipeline.scriptorium.artifacts` is defined.
|
import, profile, default, campaign, party, or family source without printing
|
||||||
|
the field value or credential contents.
|
||||||
|
|
||||||
Links:
|
Safe fix:
|
||||||
- [docs/cli.md](./cli.md)
|
|
||||||
- [docs/config.md](./config.md)
|
|
||||||
|
|
||||||
## `run-stage --artifacts` on non-analyze stage
|
- move a duplicated base field so it has one owner;
|
||||||
|
- correct the selected profile or its overlay; or
|
||||||
|
- correct the campaign party/family declaration that owns generated values.
|
||||||
|
|
||||||
|
To review what would actually change before switching profiles, run `config
|
||||||
|
diff` with the same pipeline and campaign selectors. It compares normalized
|
||||||
|
effective values rather than YAML formatting or source-file layout.
|
||||||
|
|
||||||
|
Relevant reference: [Configuration inspection](./config.md#read-only-effective-pipeline-inspection).
|
||||||
|
|
||||||
|
## Audio mode conflict
|
||||||
|
|
||||||
Symptom:
|
Symptom:
|
||||||
- `run-stage` fails with `--artifacts is only supported for stage "analyze"`.
|
|
||||||
|
|
||||||
Likely Cause:
|
- validation fails on session audio configuration.
|
||||||
- `--artifacts` was used with a non-`analyze` stage.
|
|
||||||
|
Likely cause:
|
||||||
|
|
||||||
|
- configured both local and S3 session audio inputs.
|
||||||
|
|
||||||
Diagnostics:
|
Diagnostics:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio run-stage --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 --artifacts session_recap polish
|
narratio session validate 2026-04-04
|
||||||
```
|
```
|
||||||
|
|
||||||
Safe Fix:
|
Safe fix:
|
||||||
- use `--artifacts` only with `run-stage ... analyze`.
|
|
||||||
|
|
||||||
Links:
|
- use local mode (`audio_dir` or `audio_files`) or S3 mode (`audio_s3.prefix`), not both.
|
||||||
- [docs/cli.md](./cli.md)
|
|
||||||
|
|
||||||
## Configured artifact dependency/input validation failure
|
Relevant reference: [Session configuration](./config.md#session).
|
||||||
|
|
||||||
|
## `--artifacts` selection error
|
||||||
|
|
||||||
Symptom:
|
Symptom:
|
||||||
- config validation fails for `depends_on`, `narratio.artifact.<name>` source, or artifact output path.
|
|
||||||
|
|
||||||
Likely Cause:
|
- unknown artifact key or invalid `--artifacts` usage.
|
||||||
- `narratio.artifact.<name>` source missing matching `depends_on` key.
|
|
||||||
- dependency references unknown artifact key.
|
Likely causes:
|
||||||
- dependency self-reference or enabled dependency cycle.
|
|
||||||
- artifact output path missing/invalid/outside `artifacts/` root.
|
- key not defined in `pipeline.scriptorium.artifacts`;
|
||||||
|
- empty list entry (for example trailing comma);
|
||||||
|
- `run-stage` used with non-`analyze`/`publish` target.
|
||||||
|
|
||||||
Diagnostics:
|
Diagnostics:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio plan --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04
|
narratio session artifacts 2026-04-04
|
||||||
```
|
```
|
||||||
|
|
||||||
Safe Fix:
|
Safe fix:
|
||||||
- ensure artifact-to-artifact inputs have explicit `depends_on` entries using artifact keys.
|
|
||||||
- ensure referenced artifacts exist and define valid `output_path` values.
|
|
||||||
- keep output paths relative and under `artifacts/`.
|
|
||||||
|
|
||||||
Links:
|
- provide only configured keys and use `--artifacts` with supported commands/stages.
|
||||||
- [docs/config.md](./config.md)
|
|
||||||
- [docs/internal/stage-analyze.md](./internal/stage-analyze.md)
|
|
||||||
|
|
||||||
## Required configured artifact input unavailable at analyze time
|
Relevant reference: [CLI artifact selection](./cli.md).
|
||||||
|
|
||||||
|
## Bounded run prerequisite is unusable
|
||||||
|
|
||||||
Symptom:
|
Symptom:
|
||||||
- analyze fails because configured input source is unavailable.
|
|
||||||
|
|
||||||
Likely Cause:
|
- `run` or `session plan` reports that a prerequisite stage is absent or has a
|
||||||
- required upstream configured artifact was not selected/executed this run.
|
pending, running, failed, stale, or interrupted status before the selected
|
||||||
- non-executable dependency output file is missing or invalid on disk.
|
start.
|
||||||
|
|
||||||
|
Likely cause:
|
||||||
|
|
||||||
|
- `--from` excludes upstream work that has not reached the terminal
|
||||||
|
`succeeded` or `skipped` state in the session manifest.
|
||||||
|
|
||||||
Diagnostics:
|
Diagnostics:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio status --manifest /path/to/manifest.json
|
narratio session status 2026-04-04
|
||||||
narratio run-stage --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 --artifacts player_handout analyze
|
narratio session plan 2026-04-04 --from render --through analyze
|
||||||
```
|
```
|
||||||
|
|
||||||
Safe Fix:
|
Safe fix:
|
||||||
- run analyze with needed artifacts selected.
|
|
||||||
- or ensure dependency output file exists at configured path and is valid.
|
|
||||||
|
|
||||||
Links:
|
- widen the bounded range to include the first reported stage, or recover that
|
||||||
- [docs/operations.md](./operations.md)
|
stage explicitly with `run-stage` before retrying. The failed check does not
|
||||||
- [docs/config.md](./config.md)
|
create a run record or modify the manifest. Narratio does not resume-validate
|
||||||
|
excluded prefix stages, and stages after `--through` are not prerequisites.
|
||||||
|
|
||||||
## Manifest/status path failure
|
If prerequisite statuses are terminal but a selected stage reports a missing,
|
||||||
|
unsafe, or checksum-inconsistent artifact, repair the artifact at the stage
|
||||||
|
that owns it; do not edit the manifest to bypass the selected stage's concrete
|
||||||
|
input validation.
|
||||||
|
|
||||||
|
Relevant reference: [Operations: Stage Execution and Continuation Behavior](./operations.md#stage-execution-and-continuation-behavior).
|
||||||
|
|
||||||
|
## Notarius executable missing
|
||||||
|
|
||||||
Symptom:
|
Symptom:
|
||||||
- `status` fails because manifest path is missing, unreadable, or invalid.
|
|
||||||
|
|
||||||
Likely Cause:
|
- extraction fails while resolving or starting the Notarius executable.
|
||||||
- wrong manifest path.
|
|
||||||
- manifest removed after cleanup.
|
Likely causes:
|
||||||
- `--manifest` omitted.
|
|
||||||
|
- `pipeline.notarius.binary` is not installed, executable, or on `PATH`;
|
||||||
|
- a configured executable path is wrong.
|
||||||
|
|
||||||
|
Safe fix:
|
||||||
|
|
||||||
|
- install a compatible Notarius release or correct the binary setting, then
|
||||||
|
rerun extraction.
|
||||||
|
|
||||||
|
Relevant references: [Notarius configuration](./config.md#notarius-output-entries)
|
||||||
|
and [Notarius integration](./integrations/notarius.md).
|
||||||
|
|
||||||
|
## Notarius exits nonzero
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
|
||||||
|
- extraction reports a Notarius exit error instead of a receipt.
|
||||||
|
|
||||||
|
Diagnostics:
|
||||||
|
|
||||||
|
- inspect `runs/{run_id}/extract/notarius.stderr.log`; stdout is reserved for
|
||||||
|
the receipt and is not merged with diagnostics.
|
||||||
|
|
||||||
|
Safe fix:
|
||||||
|
|
||||||
|
- correct the reported Notarius pipeline, input, provider, or configuration
|
||||||
|
failure and rerun extraction. Do not edit a staged output bundle into place.
|
||||||
|
|
||||||
|
After a failed replacement, an older immutable bundle may still exist even
|
||||||
|
though the current session manifest has no successful extraction payload. This
|
||||||
|
is expected audit state, not a signal to relink the old bundle manually.
|
||||||
|
|
||||||
|
Relevant reference: [Operations: Extraction Workflow](./operations.md#extraction-workflow).
|
||||||
|
|
||||||
|
## Prepared Notarius reference missing or inconsistent
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
|
||||||
|
- extraction or resume validation reports that a configured reference source is
|
||||||
|
unavailable, unsafe, empty, or checksum-inconsistent and recommends
|
||||||
|
`prepare --force`.
|
||||||
|
|
||||||
|
Likely causes:
|
||||||
|
|
||||||
|
- `prepare` has not run since the campaign/session stable input changed;
|
||||||
|
- the configured source file is missing;
|
||||||
|
- a prepared `inputs/` file or its manifest record was modified independently;
|
||||||
|
- a spell-catalog binding exists without an effective `spell_catalog_file`.
|
||||||
|
|
||||||
Diagnostics:
|
Diagnostics:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio status --manifest /path/to/manifest.json
|
narratio session status 2026-04-04
|
||||||
ls -l /path/to/manifest.json
|
narratio session validate 2026-04-04
|
||||||
```
|
```
|
||||||
|
|
||||||
Safe Fix:
|
Safe fix:
|
||||||
- use manifest path printed by `run`, `resume`, or `run-stage`.
|
|
||||||
|
|
||||||
Links:
|
- correct the campaign/session input path, then refresh canonical prepared
|
||||||
- [docs/cli.md](./cli.md)
|
evidence before extraction:
|
||||||
- [docs/operations.md](./operations.md)
|
|
||||||
|
```bash
|
||||||
|
narratio run-stage prepare 2026-04-04 --force
|
||||||
|
```
|
||||||
|
|
||||||
|
Do not point Notarius directly at the original source path or edit the manifest
|
||||||
|
checksum. Relevant references: [Notarius reference configuration](./config.md#notarius-reference-bindings)
|
||||||
|
and [Operations: Extraction Workflow](./operations.md#extraction-workflow).
|
||||||
|
|
||||||
|
## Notarius reference selector or generated-handoff collision
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
|
||||||
|
- Notarius exits nonzero with an undeclared reference-slot, incompatible media,
|
||||||
|
or external/generated reference collision error.
|
||||||
|
|
||||||
|
Likely causes:
|
||||||
|
|
||||||
|
- a selector does not identify a slot declared by the selected Notarius target;
|
||||||
|
- a prepared file does not satisfy that slot's Notarius media contract; or
|
||||||
|
- a CLI binding attempts to replace a same-run generated D&D handoff.
|
||||||
|
|
||||||
|
Safe fix:
|
||||||
|
|
||||||
|
- compare external bindings with the selected Notarius pipeline's canonical
|
||||||
|
consumer documentation;
|
||||||
|
- keep only campaign-owned external slots on the CLI; and
|
||||||
|
- leave registry, scene, combat, and occurrence handoffs to Notarius pipeline
|
||||||
|
composition.
|
||||||
|
|
||||||
|
Narratio validates selector structure and prepared evidence, while Notarius
|
||||||
|
owns slot declarations, media compatibility, and generated-handoff conflicts.
|
||||||
|
Relevant reference: [Notarius integration](./integrations/notarius.md).
|
||||||
|
|
||||||
|
## Atomic Notarius promotion unsupported
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
|
||||||
|
- extraction fails with `atomic no-replace directory promotion is unsupported`
|
||||||
|
before a durable bundle or temporary promotion tree is created.
|
||||||
|
|
||||||
|
Likely cause:
|
||||||
|
|
||||||
|
- Narratio is running on an operating system other than Linux, macOS, or
|
||||||
|
Windows, where the required atomic no-replace directory primitive has not
|
||||||
|
been implemented and verified.
|
||||||
|
|
||||||
|
Safe fix:
|
||||||
|
|
||||||
|
- run extraction on Linux, macOS, or Windows. Do not replace the atomic commit
|
||||||
|
with a manual copy or move; the session manifest must never observe a partial
|
||||||
|
or overwritten bundle.
|
||||||
|
|
||||||
|
This is an extraction-specific platform boundary, not a support statement for
|
||||||
|
unrelated Narratio workflows. See
|
||||||
|
[Operations: Extraction Workflow](./operations.md#extraction-workflow).
|
||||||
|
|
||||||
|
## Notarius receipt or index incompatible
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
|
||||||
|
- extraction rejects the receipt schema, pipeline identity, bundle/index path,
|
||||||
|
lane descriptor, or payload path even though Notarius exited successfully.
|
||||||
|
|
||||||
|
Likely causes:
|
||||||
|
|
||||||
|
- Narratio and Notarius versions disagree on their consumer contract;
|
||||||
|
- the configured pipeline or lane constraints are stale;
|
||||||
|
- output paths escape the bundle or traverse symlinks.
|
||||||
|
|
||||||
|
Safe fix:
|
||||||
|
|
||||||
|
- compare installed Notarius output with the canonical Notarius contracts,
|
||||||
|
including receipt `index_file: index.json` and index management names
|
||||||
|
`manifest.json`, `rejected.json`, `warnings.json`, and `diagnostics.json`; align
|
||||||
|
`pipeline.notarius` constraints and rerun. Do not bypass confinement or schema
|
||||||
|
checks.
|
||||||
|
|
||||||
|
Relevant reference: [Notarius integration](./integrations/notarius.md).
|
||||||
|
|
||||||
|
## Required Notarius lane rejected or missing
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
|
||||||
|
- extraction fails because a configured lane is rejected, missing, duplicated,
|
||||||
|
or incompatible, including after a zero exit.
|
||||||
|
|
||||||
|
Safe fix:
|
||||||
|
|
||||||
|
- inspect the Notarius diagnostic log and bundle rejection/warning information;
|
||||||
|
- correct the Notarius module or the exact declared lane contract;
|
||||||
|
- remove an output declaration only if downstream consumers genuinely no longer
|
||||||
|
require that source, then rerun extraction.
|
||||||
|
|
||||||
|
Every configured output is required. Narratio does not promote a partial result.
|
||||||
|
|
||||||
|
## Extraction resume invalidated
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
|
||||||
|
- a previously successful extraction runs again during ordinary continuation.
|
||||||
|
|
||||||
|
Likely causes:
|
||||||
|
|
||||||
|
- the executable/config path, pipeline ID, timeout, working directory, or
|
||||||
|
configured output contracts changed;
|
||||||
|
- a configured prepared reference selector, source, path, checksum, or byte
|
||||||
|
size changed;
|
||||||
|
- the durable bundle, index, lane set, provenance, regular-file status, or
|
||||||
|
checksum no longer validates.
|
||||||
|
|
||||||
|
Safe fix:
|
||||||
|
|
||||||
|
- allow the automatic rerun after verifying the current configuration. Treat
|
||||||
|
an unsafe path or symlink error as filesystem corruption or tampering and
|
||||||
|
investigate it rather than replacing files manually.
|
||||||
|
|
||||||
|
## Notarius transitive configuration changed
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
|
||||||
|
- Notarius profiles, prompts, modules, imported files, or references changed,
|
||||||
|
but Narratio still considers the previous extraction resumable.
|
||||||
|
|
||||||
|
Safe fix:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run-stage extract 2026-04-04 --force
|
||||||
|
```
|
||||||
|
|
||||||
|
Narratio fingerprints its invocation contract and prepared Narratio reference
|
||||||
|
identities, not the contents of other transitive Notarius inputs. Always force
|
||||||
|
extraction after changing those external inputs; downstream
|
||||||
|
successful stages are then marked stale normally.
|
||||||
|
|
||||||
|
Relevant reference: [Operations: Extraction Workflow](./operations.md#extraction-workflow).
|
||||||
|
|
||||||
|
## Analysis artifact evidence is not current
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
|
||||||
|
- ordinary continuation or `session plan` schedules one or more configured
|
||||||
|
artifacts even though a canonical output file exists; or
|
||||||
|
- publish reports a configured artifact source unavailable.
|
||||||
|
|
||||||
|
Likely causes:
|
||||||
|
|
||||||
|
- the per-artifact record is stale, missing, failed, unselected, malformed, or
|
||||||
|
from the legacy aggregate-only manifest contract;
|
||||||
|
- a configured prompt/profile, dependency, input identity, output path, or
|
||||||
|
effective variable changed; or
|
||||||
|
- the recorded output is missing, unsafe, empty, or has a size/checksum that no
|
||||||
|
longer matches its manifest evidence.
|
||||||
|
|
||||||
|
Diagnostics:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session status 2026-04-04
|
||||||
|
narratio session artifacts 2026-04-04
|
||||||
|
narratio session plan 2026-04-04 --from analyze --through analyze
|
||||||
|
```
|
||||||
|
|
||||||
|
Safe fix:
|
||||||
|
|
||||||
|
- investigate unexpected path or checksum changes as possible tampering;
|
||||||
|
- otherwise let the selected analyze work rerun, or explicitly regenerate only
|
||||||
|
the affected targets; and
|
||||||
|
- never edit the fingerprint/checksum in the manifest or copy an old file into
|
||||||
|
the canonical path as a substitute for current evidence.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio analyze 2026-04-04 --artifacts session_recap
|
||||||
|
```
|
||||||
|
|
||||||
|
Relevant references: [Operations: Artifact Selection](./operations.md#artifact-selection)
|
||||||
|
and [Artifact Internals](./internal/artifacts.md#resolution-rules).
|
||||||
|
|
||||||
|
## Legacy aggregate analysis requires regeneration
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
|
||||||
|
- a manifest from an older Narratio version reports aggregate analyze success
|
||||||
|
and the old files are present, but configured artifact sources remain
|
||||||
|
unavailable.
|
||||||
|
|
||||||
|
Likely cause:
|
||||||
|
|
||||||
|
- the manifest has no supported per-artifact analyze state. Aggregate output
|
||||||
|
lists do not establish current configured-artifact authority.
|
||||||
|
|
||||||
|
Safe fix:
|
||||||
|
|
||||||
|
- regenerate the required artifacts. A partial selection makes only its
|
||||||
|
targets and prerequisites eligible for current state; unselected legacy
|
||||||
|
files intentionally remain unavailable. Run full analysis later when every
|
||||||
|
enabled configured artifact must become current.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio analyze 2026-04-04 --artifacts session_recap
|
||||||
|
narratio analyze 2026-04-04
|
||||||
|
```
|
||||||
|
|
||||||
|
After current records exist, inspect them and publish explicitly. Do not delete
|
||||||
|
the legacy files merely to influence selection; availability is manifest-owned.
|
||||||
|
|
||||||
|
Relevant references: [Operations: Stage Execution and Continuation Behavior](./operations.md#stage-execution-and-continuation-behavior)
|
||||||
|
and [Manifest Internals](./internal/manifest.md#analyze-owned-artifact-state).
|
||||||
|
|
||||||
|
## Scriptorium private input changed without a rerun
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
|
||||||
|
- a prompt, profile, imported configuration file, executable, or other input
|
||||||
|
loaded privately by Scriptorium changed, but Narratio still considers an
|
||||||
|
artifact current.
|
||||||
|
|
||||||
|
Likely cause:
|
||||||
|
|
||||||
|
- analysis fingerprints cover Narratio-observable semantic identities, not
|
||||||
|
executable contents or arbitrary files and transitive configuration that
|
||||||
|
Scriptorium loads behind its configured paths and identifiers.
|
||||||
|
|
||||||
|
Safe fix:
|
||||||
|
|
||||||
|
- explicitly force the affected target after changing an unobserved private
|
||||||
|
input. Force applies to explicit targets; current prerequisites remain
|
||||||
|
reusable unless selected themselves.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio analyze 2026-04-04 --artifacts session_recap
|
||||||
|
```
|
||||||
|
|
||||||
|
Relevant reference: [Analyze Internals](./internal/stage-analyze.md#invariants).
|
||||||
|
|
||||||
|
## Previous-session artifact input missing
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
|
||||||
|
- prepare/analyze fails due to missing required previous-session artifact cache input.
|
||||||
|
|
||||||
|
Likely causes:
|
||||||
|
|
||||||
|
- missing `session.previous_session_id`;
|
||||||
|
- previous artifact not restored/published for source session.
|
||||||
|
|
||||||
|
Diagnostics:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session validate 2026-04-04
|
||||||
|
narratio session status 2026-04-04
|
||||||
|
```
|
||||||
|
|
||||||
|
Safe fix:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session restore 2026-04-04
|
||||||
|
```
|
||||||
|
|
||||||
|
or rerun prepare after correcting session config:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run-stage prepare 2026-04-04 --force
|
||||||
|
```
|
||||||
|
|
||||||
|
Relevant reference: [Operations: Restore Workflow](./operations.md#restore-workflow).
|
||||||
|
|
||||||
## Session lock conflict (`.lock`)
|
## Session lock conflict (`.lock`)
|
||||||
|
|
||||||
Symptom:
|
Symptom:
|
||||||
- run fails with lock conflict for session workdir.
|
|
||||||
|
|
||||||
Likely Cause:
|
- command fails acquiring session lock.
|
||||||
- another Narratio process is running same session.
|
|
||||||
- stale lock from interrupted prior run.
|
Likely causes:
|
||||||
|
|
||||||
|
- another process is running for the same session;
|
||||||
|
- a process still holds the operating-system lock while it is shutting down.
|
||||||
|
|
||||||
Diagnostics:
|
Diagnostics:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
ls -l {workspace.root}/work/{campaign}/{session_id}/.lock
|
ls -l {workspace.root}/work/{campaign}/{session_id}/.lock
|
||||||
cat {workspace.root}/work/{campaign}/{session_id}/.lock
|
|
||||||
ps aux | grep narratio
|
ps aux | grep narratio
|
||||||
```
|
```
|
||||||
|
|
||||||
Safe Fix:
|
Safe fix:
|
||||||
- wait for active run to finish.
|
|
||||||
- if no process is active, remove only stale session `.lock` file.
|
|
||||||
|
|
||||||
Links:
|
- wait for active process completion;
|
||||||
- [docs/operations.md](./operations.md)
|
- retry after an interrupted holder has exited; the kernel releases its lock
|
||||||
- [docs/internal/workspace.md](./internal/workspace.md)
|
even though the `.lock` metadata file remains for inspection.
|
||||||
|
|
||||||
## Secrets env-dir or credential-env failure
|
Relevant reference: [Operations: Local State Layout](./operations.md#local-state-layout).
|
||||||
|
|
||||||
|
## Restore conflict without `--force`
|
||||||
|
|
||||||
Symptom:
|
Symptom:
|
||||||
- startup fails loading secrets directory, or stage fails due to missing credential env vars.
|
|
||||||
|
|
||||||
Likely Cause:
|
- restore fails with conflict count.
|
||||||
- invalid `pipeline.secrets.env_dir` path/permissions.
|
|
||||||
- required credential env var unset/empty.
|
Likely cause:
|
||||||
|
|
||||||
|
- local durable files differ from remote restore sources.
|
||||||
|
|
||||||
|
Diagnostics:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session restore 2026-04-04 --dry-run
|
||||||
|
```
|
||||||
|
|
||||||
|
Safe fix:
|
||||||
|
|
||||||
|
- review conflicts;
|
||||||
|
- rerun with `--force` only when remote state should overwrite local.
|
||||||
|
|
||||||
|
Relevant reference: [Operations: Restore Workflow](./operations.md#restore-workflow).
|
||||||
|
|
||||||
|
## Restore current-state discovery failure
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
|
||||||
|
- restore cannot find current pointer or current manifest.
|
||||||
|
|
||||||
|
Likely causes:
|
||||||
|
|
||||||
|
- no committed publish current state;
|
||||||
|
- storage credentials or connectivity failure.
|
||||||
|
|
||||||
|
Diagnostics:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session status 2026-04-04
|
||||||
|
narratio session restore 2026-04-04 --dry-run
|
||||||
|
```
|
||||||
|
|
||||||
|
Safe fix:
|
||||||
|
|
||||||
|
- resolve storage/auth issue;
|
||||||
|
- republish from healthy local state if current pointer is missing.
|
||||||
|
|
||||||
|
Relevant reference: [Operations: Publish Workflow](./operations.md#publish-workflow).
|
||||||
|
|
||||||
|
## Publish output failure
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
|
||||||
|
- publish fails on missing required source, upload error, or commit write.
|
||||||
|
|
||||||
|
Likely causes:
|
||||||
|
|
||||||
|
- required source file not produced;
|
||||||
|
- lock/state expectations mismatch;
|
||||||
|
- remote storage failure.
|
||||||
|
|
||||||
|
Diagnostics:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session artifacts 2026-04-04 --remote
|
||||||
|
narratio session status 2026-04-04
|
||||||
|
narratio run-stage publish 2026-04-04 --force
|
||||||
|
```
|
||||||
|
|
||||||
|
Safe fix:
|
||||||
|
|
||||||
|
- regenerate missing sources by rerunning prerequisite stages;
|
||||||
|
- correct publish source/destination rules;
|
||||||
|
- retry after storage failure is resolved.
|
||||||
|
|
||||||
|
Relevant reference: [Publish configuration](./config.md#publish-configuration-summary).
|
||||||
|
|
||||||
|
## Render markdown source missing
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
|
||||||
|
- analyze or publish fails because `narratio.transcript.final_markdown` or `narratio.transcript.final_trimmed_markdown` is unavailable.
|
||||||
|
|
||||||
|
Likely causes:
|
||||||
|
|
||||||
|
- render stage was not executed after transcript changes;
|
||||||
|
- render stage failed before producing canonical markdown outputs.
|
||||||
|
|
||||||
|
Diagnostics:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio session status 2026-04-04
|
||||||
|
```
|
||||||
|
|
||||||
|
Safe fix:
|
||||||
|
|
||||||
|
- rerun render and then retry downstream stage(s):
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run-stage render 2026-04-04 --force
|
||||||
|
narratio run-stage analyze 2026-04-04 --force
|
||||||
|
```
|
||||||
|
|
||||||
|
Relevant reference: [Operations: Stage Execution](./operations.md#stage-execution-and-continuation-behavior).
|
||||||
|
|
||||||
|
## Secrets or storage credential failure
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
|
||||||
|
- object-store command fails at initialization/auth.
|
||||||
|
|
||||||
|
Likely causes:
|
||||||
|
|
||||||
|
- invalid `pipeline.secrets.env_dir`;
|
||||||
|
- missing credential environment variables;
|
||||||
|
- invalid S3 endpoint/bucket settings.
|
||||||
|
|
||||||
Diagnostics:
|
Diagnostics:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
ls -la /path/to/secrets_dir
|
ls -la /path/to/secrets_dir
|
||||||
env | grep -E 'AUDITA|OBJECT_STORAGE|AWS|SCRIPTORIUM'
|
env | sed 's/=.*//' | grep -E 'OBJECT_STORAGE|AWS|AUDITA|SCRIPTORIUM'
|
||||||
```
|
```
|
||||||
|
|
||||||
Safe Fix:
|
Safe fix:
|
||||||
- fix secrets directory and credential env vars.
|
|
||||||
|
- correct secret-file path and permissions;
|
||||||
|
- provide required env vars;
|
||||||
- keep secret values out of YAML.
|
- keep secret values out of YAML.
|
||||||
|
|
||||||
Links:
|
Relevant reference: [Secrets](./config.md#secrets-handling).
|
||||||
- [docs/config.md](./config.md)
|
|
||||||
|
|
||||||
## S3-audio prepare failure
|
## S3 audio prepare failure
|
||||||
|
|
||||||
Symptom:
|
Symptom:
|
||||||
- `prepare` fails in S3 mode (listing/downloading/no audio/backend error).
|
|
||||||
|
|
||||||
Likely Cause:
|
- prepare fails listing/downloading session S3 audio.
|
||||||
- wrong `session.inputs.audio_s3.prefix`.
|
|
||||||
- no `.flac` files at resolved prefix.
|
Likely causes:
|
||||||
- invalid/missing object-store credentials or backend config.
|
|
||||||
- mixed local+S3 audio input config.
|
- incorrect `session.inputs.audio_s3.prefix`;
|
||||||
|
- no matching `.flac` objects;
|
||||||
|
- storage connectivity or permissions failure.
|
||||||
|
|
||||||
Diagnostics:
|
Diagnostics:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
narratio run-stage --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 prepare
|
narratio session validate 2026-04-04
|
||||||
```
|
```
|
||||||
|
|
||||||
Safe Fix:
|
Safe fix:
|
||||||
- configure exactly one audio source mode.
|
|
||||||
- verify `.flac` files and storage access.
|
|
||||||
|
|
||||||
Links:
|
- verify prefix contents and storage access;
|
||||||
|
- keep session audio mode consistent.
|
||||||
|
|
||||||
|
Relevant reference: [Operations](./operations.md).
|
||||||
|
|
||||||
|
## References
|
||||||
|
|
||||||
|
- [docs/cli.md](./cli.md)
|
||||||
- [docs/config.md](./config.md)
|
- [docs/config.md](./config.md)
|
||||||
- [docs/operations.md](./operations.md)
|
- [docs/operations.md](./operations.md)
|
||||||
|
- [docs/internal/stage-publish.md](./internal/stage-publish.md)
|
||||||
## Archive promotion/current-pointer failure
|
|
||||||
|
|
||||||
Symptom:
|
|
||||||
- archive fails on required promotion source missing or pointer write failure.
|
|
||||||
|
|
||||||
Likely Cause:
|
|
||||||
- required promoted file absent (including analyze outputs not generated for this run).
|
|
||||||
- storage upload failed before `current/run_id.txt` commit marker write.
|
|
||||||
|
|
||||||
Diagnostics:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
narratio status --manifest /path/to/manifest.json
|
|
||||||
narratio run-stage --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 archive
|
|
||||||
```
|
|
||||||
|
|
||||||
Safe Fix:
|
|
||||||
- rerun or resume upstream stages to generate required files.
|
|
||||||
- adjust promotion rules to match files that must exist.
|
|
||||||
- retry after storage issue is resolved.
|
|
||||||
|
|
||||||
Links:
|
|
||||||
- [docs/operations.md](./operations.md)
|
|
||||||
- [docs/config.md](./config.md)
|
|
||||||
- [docs/internal/stage-archive.md](./internal/stage-archive.md)
|
|
||||||
|
|||||||
62
examples/README.md
Normal file
62
examples/README.md
Normal file
@@ -0,0 +1,62 @@
|
|||||||
|
# Maintained Examples
|
||||||
|
|
||||||
|
These files are safe, copyable starting points for Narratio configuration and
|
||||||
|
input structure. Replace placeholder identifiers, storage names, integration
|
||||||
|
URLs, and paths for the target environment. Field meanings and defaults belong
|
||||||
|
in the [configuration reference](../docs/config.md).
|
||||||
|
|
||||||
|
## Pipeline Configuration
|
||||||
|
|
||||||
|
- [Minimal pipeline](pipeline.minimal.yml): campaign discovery plus the required
|
||||||
|
WhisperX URL.
|
||||||
|
- [Production-shaped pipeline](pipeline.production.yml): S3 storage, publish,
|
||||||
|
external tools, and configured Scriptorium artifacts.
|
||||||
|
- [Production/testing split bundle](production-testing/pipeline.yml): explicit
|
||||||
|
`conf.d` imports, a production default, and selectable production/testing
|
||||||
|
overlays. It also demonstrates canonical-party artifact families and a
|
||||||
|
testing-only disabled artifact. Validate it with:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
narratio config validate --config examples/production-testing/pipeline.yml --campaign-file examples/campaigns/sample-campaign/campaign.yml
|
||||||
|
narratio config diff production testing --config examples/production-testing/pipeline.yml --campaign-file examples/campaigns/sample-campaign/campaign.yml
|
||||||
|
```
|
||||||
|
|
||||||
|
`config show` and `config sources` accept the same selectors and remain
|
||||||
|
read-only.
|
||||||
|
- [Full annotated pipeline](pipeline.full.annotated.yml): every implemented
|
||||||
|
pipeline section with explanatory comments.
|
||||||
|
- [Extraction subset pipeline](pipeline.extraction-subset.yml): a focused
|
||||||
|
Scriptorium artifact consuming only three declared Notarius lanes.
|
||||||
|
|
||||||
|
The existing `internal/config` example test loads and validates each pipeline
|
||||||
|
with the sample campaign and a compatible local- or S3-audio session.
|
||||||
|
|
||||||
|
## Campaign And Session Configuration
|
||||||
|
|
||||||
|
- [Sample campaign](campaigns/sample-campaign/campaign.yml), its
|
||||||
|
[session template](campaigns/sample-campaign/session.template.yml), and its
|
||||||
|
adjacent stable inputs provide a complete campaign directory shape.
|
||||||
|
- [Local-audio session](session.local-audio.yml) and
|
||||||
|
[S3-audio session](session.s3-audio.yml) are concrete session files.
|
||||||
|
- [Session template](session.template.yml) and the campaign-local equivalent
|
||||||
|
demonstrate the narrow placeholder syntax consumed by `session init`; they
|
||||||
|
are templates, not runtime session files.
|
||||||
|
|
||||||
|
## Input Fixtures
|
||||||
|
|
||||||
|
- [Speakers](speakers.yml), [autocorrect](autocorrect.yml), and
|
||||||
|
[glossary](glossary.yml) show the standalone input shapes.
|
||||||
|
- The sample campaign references its local
|
||||||
|
[speakers](campaigns/sample-campaign/speakers.yml),
|
||||||
|
[autocorrect](campaigns/sample-campaign/autocorrect.yml),
|
||||||
|
[glossary](campaigns/sample-campaign/glossary.yml), and canonical
|
||||||
|
[party](campaigns/sample-campaign/party.yml) fixture, plus an optional
|
||||||
|
[spell-catalog overlay](campaigns/sample-campaign/spell_catalog.json) that
|
||||||
|
follows the Notarius v0.6 contract. Narratio derives the players projection
|
||||||
|
from this party source; the campaign deliberately has no `players_file`.
|
||||||
|
- [Sample speaker audio](audio/sample-speaker.flac) is a text placeholder that
|
||||||
|
reserves the expected filename and directory shape. Replace it with a real
|
||||||
|
FLAC file before running transcription.
|
||||||
|
|
||||||
|
The examples contain environment-variable names but no credential values. They
|
||||||
|
use fictional campaign content and reserved example domains.
|
||||||
1
examples/campaigns/sample-campaign/autocorrect.yml
Normal file
1
examples/campaigns/sample-campaign/autocorrect.yml
Normal file
@@ -0,0 +1 @@
|
|||||||
|
[]
|
||||||
8
examples/campaigns/sample-campaign/campaign.yml
Normal file
8
examples/campaigns/sample-campaign/campaign.yml
Normal file
@@ -0,0 +1,8 @@
|
|||||||
|
campaign_id: sample-campaign
|
||||||
|
session_template_file: ./session.template.yml
|
||||||
|
inputs:
|
||||||
|
speakers_file: ./speakers.yml
|
||||||
|
autocorrect_file: ./autocorrect.yml
|
||||||
|
glossary_file: ./glossary.yml
|
||||||
|
party_file: ./party.yml
|
||||||
|
spell_catalog_file: ./spell_catalog.json
|
||||||
1
examples/campaigns/sample-campaign/glossary.yml
Normal file
1
examples/campaigns/sample-campaign/glossary.yml
Normal file
@@ -0,0 +1 @@
|
|||||||
|
[]
|
||||||
26
examples/campaigns/sample-campaign/party.yml
Normal file
26
examples/campaigns/sample-campaign/party.yml
Normal file
@@ -0,0 +1,26 @@
|
|||||||
|
schema_version: narratio.party.v1
|
||||||
|
|
||||||
|
characters:
|
||||||
|
arannis:
|
||||||
|
player:
|
||||||
|
name: Rowan Hale
|
||||||
|
character:
|
||||||
|
name: Arannis
|
||||||
|
alias:
|
||||||
|
- Ari
|
||||||
|
- The Grey Owl
|
||||||
|
classes:
|
||||||
|
- name: wizard
|
||||||
|
level: 8
|
||||||
|
brenna:
|
||||||
|
player:
|
||||||
|
name: Rowan Hale
|
||||||
|
character:
|
||||||
|
name: Brenna
|
||||||
|
alias:
|
||||||
|
- Shield of Dawn
|
||||||
|
classes:
|
||||||
|
- name: paladin
|
||||||
|
level: 6
|
||||||
|
- name: warlock
|
||||||
|
level: 2
|
||||||
3
examples/campaigns/sample-campaign/session.template.yml
Normal file
3
examples/campaigns/sample-campaign/session.template.yml
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
session_id: "{{ session_id }}"
|
||||||
|
inputs:
|
||||||
|
audio_dir: ./audio
|
||||||
5
examples/campaigns/sample-campaign/speakers.yml
Normal file
5
examples/campaigns/sample-campaign/speakers.yml
Normal file
@@ -0,0 +1,5 @@
|
|||||||
|
match:
|
||||||
|
- speaker: "Example Speaker"
|
||||||
|
match:
|
||||||
|
- "Example_Speaker"
|
||||||
|
- "Example"
|
||||||
18
examples/campaigns/sample-campaign/spell_catalog.json
Normal file
18
examples/campaigns/sample-campaign/spell_catalog.json
Normal file
@@ -0,0 +1,18 @@
|
|||||||
|
{
|
||||||
|
"schema_version": "notarius.dnd.spell-catalog-overlay.v1",
|
||||||
|
"catalogs": [
|
||||||
|
{
|
||||||
|
"id": "narratio.sample-campaign",
|
||||||
|
"ruleset": "dnd-5e-2014",
|
||||||
|
"source": {
|
||||||
|
"title": "Narratio sample campaign spell names"
|
||||||
|
},
|
||||||
|
"spells": [
|
||||||
|
{
|
||||||
|
"name": "Aegis of Emberfall",
|
||||||
|
"aliases": ["Emberfall Aegis"]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
59
examples/pipeline.extraction-subset.yml
Normal file
59
examples/pipeline.extraction-subset.yml
Normal file
@@ -0,0 +1,59 @@
|
|||||||
|
# Purpose-specific extraction example: a Scriptorium session brief consumes
|
||||||
|
# only the three Notarius lanes it needs.
|
||||||
|
|
||||||
|
campaigns:
|
||||||
|
root: /usr/local/share/narratio/campaigns
|
||||||
|
default_campaign_id: sample-campaign
|
||||||
|
|
||||||
|
whisperx:
|
||||||
|
transcribe_url: "https://transcription.example.com/transcribe"
|
||||||
|
|
||||||
|
notarius:
|
||||||
|
enabled: true
|
||||||
|
binary: notarius
|
||||||
|
config_path: /usr/local/etc/notarius/config.yml
|
||||||
|
pipeline_id: dnd-session
|
||||||
|
timeout: 3h
|
||||||
|
references:
|
||||||
|
glossary: narratio.input.glossary
|
||||||
|
party: narratio.input.party
|
||||||
|
players: narratio.input.players
|
||||||
|
spell_catalog: narratio.input.spell_catalog
|
||||||
|
outputs:
|
||||||
|
npc_registry:
|
||||||
|
lane_id: npc-registry
|
||||||
|
media_type: application/json
|
||||||
|
schema_id: notarius.dnd.npc_registry
|
||||||
|
schema_version: v1
|
||||||
|
module_key: dnd/npc-registry
|
||||||
|
location_registry:
|
||||||
|
lane_id: location-registry
|
||||||
|
media_type: application/json
|
||||||
|
schema_id: notarius.dnd.location_registry
|
||||||
|
schema_version: v1
|
||||||
|
module_key: dnd/location-registry
|
||||||
|
scene_descriptions:
|
||||||
|
lane_id: scene-descriptions
|
||||||
|
media_type: application/json
|
||||||
|
schema_id: notarius.dnd.scene_descriptions
|
||||||
|
schema_version: v1
|
||||||
|
module_key: dnd/scene-descriptions
|
||||||
|
|
||||||
|
scriptorium:
|
||||||
|
binary: scriptorium
|
||||||
|
config_path: /usr/local/etc/scriptorium/config.yml
|
||||||
|
artifacts:
|
||||||
|
session_brief:
|
||||||
|
enabled: true
|
||||||
|
prompt_id: dnd.session_brief
|
||||||
|
output_path: artifacts/session_brief.md
|
||||||
|
inputs:
|
||||||
|
npcs:
|
||||||
|
source: narratio.extraction.npc_registry
|
||||||
|
required: true
|
||||||
|
locations:
|
||||||
|
source: narratio.extraction.location_registry
|
||||||
|
required: true
|
||||||
|
scenes:
|
||||||
|
source: narratio.extraction.scene_descriptions
|
||||||
|
required: true
|
||||||
@@ -4,21 +4,18 @@
|
|||||||
workspace:
|
workspace:
|
||||||
# Optional: defaults to /var/lib/narratio.
|
# Optional: defaults to /var/lib/narratio.
|
||||||
root: /var/lib/narratio/workspace
|
root: /var/lib/narratio/workspace
|
||||||
# Optional: remove run-scoped workdir after successful archive commit.
|
# Optional: remove run-scoped workdir after successful publish commit.
|
||||||
cleanup_after_archive: false
|
cleanup_after_publish: false
|
||||||
|
|
||||||
# Optional: local secret file loader (directory of ENV_VAR_NAME files).
|
# Optional: local secret file loader (directory of ENV_VAR_NAME files).
|
||||||
# secrets:
|
# secrets:
|
||||||
# env_dir: ./secrets
|
# env_dir: ./secrets
|
||||||
|
|
||||||
storage:
|
storage:
|
||||||
# Optional storage backend selector; use "s3" for archive + S3 audio workflows.
|
# Defaults to "local". Use "s3" explicitly for publish + S3 audio workflows.
|
||||||
backend: s3
|
backend: s3
|
||||||
# Compatibility fields retained in schema.
|
|
||||||
bucket: ""
|
|
||||||
prefix: ""
|
|
||||||
s3:
|
s3:
|
||||||
# Required when using S3 audio or S3 archive uploads.
|
# Required when using S3 audio or S3 publish uploads.
|
||||||
bucket: my-dnd-archive
|
bucket: my-dnd-archive
|
||||||
# Optional; defaults to "dnd".
|
# Optional; defaults to "dnd".
|
||||||
root_prefix: dnd
|
root_prefix: dnd
|
||||||
@@ -30,27 +27,44 @@ storage:
|
|||||||
access_key_id_env: OBJECT_STORAGE_KEY_ID
|
access_key_id_env: OBJECT_STORAGE_KEY_ID
|
||||||
secret_access_key_env: OBJECT_STORAGE_KEY
|
secret_access_key_env: OBJECT_STORAGE_KEY
|
||||||
|
|
||||||
|
campaigns:
|
||||||
|
# Optional; defaults to /usr/local/share/narratio/campaigns.
|
||||||
|
root: /usr/local/share/narratio/campaigns
|
||||||
|
# Optional command default when --campaign is omitted.
|
||||||
|
default_campaign_id: sample-campaign
|
||||||
|
|
||||||
spool:
|
spool:
|
||||||
# Optional; defaults to /var/spool/narratio.
|
# Optional; defaults to /var/spool/narratio.
|
||||||
root: /var/spool/narratio
|
root: /var/spool/narratio
|
||||||
# Optional cleanup of run-scoped spool audio after successful archive commit.
|
# Optional cleanup of run-scoped spool audio after successful publish commit.
|
||||||
delete_audio_after_archive: false
|
delete_audio_after_publish: false
|
||||||
|
|
||||||
archive:
|
publish:
|
||||||
# Optional booleans; defaults are true.
|
# Optional booleans; defaults are true.
|
||||||
enabled: true
|
enabled: true
|
||||||
upload_run: true
|
upload_run: true
|
||||||
# Optional promotion rules; required files fail archive if missing.
|
# Optional publish output rules; sources use Narratio artifact source IDs.
|
||||||
promote_artifacts:
|
outputs:
|
||||||
- from: transcripts/trimmed.json
|
- source: narratio.transcript.final_trimmed
|
||||||
to: transcripts/trimmed.json
|
dest: transcripts/final.trimmed.json
|
||||||
required: true
|
required: true
|
||||||
- from: artifacts/session_recap.md
|
- source: narratio.transcript.final_markdown
|
||||||
to: artifacts/session_recap.md
|
dest: transcripts/final.md
|
||||||
required: true
|
required: true
|
||||||
- from: artifacts/player_handout.md
|
- source: narratio.transcript.final_trimmed_markdown
|
||||||
to: artifacts/player_handout.md
|
dest: transcripts/final.trimmed.md
|
||||||
|
required: true
|
||||||
|
- source: narratio.artifact.session_recap
|
||||||
|
dest: artifacts/session_recap.md
|
||||||
|
required: true
|
||||||
|
- source: narratio.artifact.player_handout
|
||||||
|
dest: artifacts/player_handout.md
|
||||||
required: false
|
required: false
|
||||||
|
# Extraction lanes publish only when named explicitly; the bundle and index
|
||||||
|
# are never implicit publish sources.
|
||||||
|
- source: narratio.extraction.npc_registry
|
||||||
|
dest: artifacts/extraction/npc-registry.json
|
||||||
|
required: true
|
||||||
|
|
||||||
whisperx:
|
whisperx:
|
||||||
# Required.
|
# Required.
|
||||||
@@ -96,25 +110,103 @@ audita:
|
|||||||
|
|
||||||
normalize:
|
normalize:
|
||||||
# Optional; defaults shown explicitly.
|
# Optional; defaults shown explicitly.
|
||||||
output_path: transcripts/normalized.json
|
output_path: transcripts/final.json
|
||||||
output_schema: seriatim-intermediate
|
output_schema: seriatim-intermediate
|
||||||
report: true
|
report: true
|
||||||
|
|
||||||
trim:
|
trim:
|
||||||
# Keep disabled unless bounds prompt integration is configured.
|
# Optional; defaults shown explicitly.
|
||||||
enabled: false
|
enabled: true
|
||||||
output_path: transcripts/trimmed.json
|
output_path: transcripts/final.trimmed.json
|
||||||
bounds:
|
bounds:
|
||||||
prompt_id: dnd.session_bounds
|
prompt_id: dnd.session_bounds
|
||||||
profile_id: local-fast
|
profile_id: ""
|
||||||
transcript_input_name: transcript
|
transcript_input_name: transcript
|
||||||
output_path: reports/session_bounds.json
|
output_path: artifacts/session_bounds.json
|
||||||
timeout: 10m
|
timeout: 10m
|
||||||
render_debug: false
|
render_debug: false
|
||||||
render_output_path: reports/session_bounds.render.json
|
|
||||||
seriatim:
|
seriatim:
|
||||||
report: false
|
report: false
|
||||||
|
|
||||||
|
notarius:
|
||||||
|
# Optional structured extraction between trim and render.
|
||||||
|
enabled: true
|
||||||
|
binary: notarius
|
||||||
|
config_path: /usr/local/etc/notarius/config.yml
|
||||||
|
pipeline_id: dnd-session
|
||||||
|
timeout: 3h
|
||||||
|
working_directory: /usr/local/etc/notarius
|
||||||
|
# External campaign references use prepared Narratio source IDs. Omit an
|
||||||
|
# optional binding when the selected Notarius pipeline does not need it.
|
||||||
|
references:
|
||||||
|
glossary: narratio.input.glossary
|
||||||
|
party: narratio.input.party
|
||||||
|
players: narratio.input.players
|
||||||
|
spell_catalog: narratio.input.spell_catalog
|
||||||
|
# Each key creates source narratio.extraction.<key>. These constraints match
|
||||||
|
# the current Notarius D&D lane contracts; update them with Notarius.
|
||||||
|
outputs:
|
||||||
|
item_registry:
|
||||||
|
lane_id: item-registry
|
||||||
|
media_type: application/json
|
||||||
|
schema_id: notarius.dnd.item_registry
|
||||||
|
schema_version: v1
|
||||||
|
module_key: dnd/item-registry
|
||||||
|
npc_registry:
|
||||||
|
lane_id: npc-registry
|
||||||
|
media_type: application/json
|
||||||
|
schema_id: notarius.dnd.npc_registry
|
||||||
|
schema_version: v1
|
||||||
|
module_key: dnd/npc-registry
|
||||||
|
location_registry:
|
||||||
|
lane_id: location-registry
|
||||||
|
media_type: application/json
|
||||||
|
schema_id: notarius.dnd.location_registry
|
||||||
|
schema_version: v1
|
||||||
|
module_key: dnd/location-registry
|
||||||
|
scene_descriptions:
|
||||||
|
lane_id: scene-descriptions
|
||||||
|
media_type: application/json
|
||||||
|
schema_id: notarius.dnd.scene_descriptions
|
||||||
|
schema_version: v1
|
||||||
|
module_key: dnd/scene-descriptions
|
||||||
|
item_occurrences:
|
||||||
|
lane_id: item-occurrences
|
||||||
|
media_type: application/json
|
||||||
|
schema_id: notarius.dnd.item_occurrences
|
||||||
|
schema_version: v1
|
||||||
|
module_key: dnd/item-occurrences
|
||||||
|
spells:
|
||||||
|
lane_id: spells
|
||||||
|
media_type: application/json
|
||||||
|
schema_id: notarius.dnd.spells
|
||||||
|
schema_version: v1
|
||||||
|
module_key: dnd/spells
|
||||||
|
combat_turns:
|
||||||
|
lane_id: combat-turns
|
||||||
|
media_type: application/json
|
||||||
|
schema_id: notarius.dnd.combat_turns
|
||||||
|
schema_version: v1
|
||||||
|
module_key: dnd/combat-turns
|
||||||
|
npc_occurrences:
|
||||||
|
lane_id: npc-occurrences
|
||||||
|
media_type: application/json
|
||||||
|
schema_id: notarius.dnd.npc_occurrences
|
||||||
|
schema_version: v1
|
||||||
|
module_key: dnd/npc-occurrences
|
||||||
|
location_occurrences:
|
||||||
|
lane_id: location-occurrences
|
||||||
|
media_type: application/json
|
||||||
|
schema_id: notarius.dnd.location_occurrences
|
||||||
|
schema_version: v1
|
||||||
|
module_key: dnd/location-occurrences
|
||||||
|
enemy_events:
|
||||||
|
lane_id: enemy-events
|
||||||
|
media_type: application/json
|
||||||
|
schema_id: notarius.dnd.enemy_events
|
||||||
|
schema_version: v1
|
||||||
|
module_key: dnd/enemy-events
|
||||||
|
|
||||||
scriptorium:
|
scriptorium:
|
||||||
binary: scriptorium
|
binary: scriptorium
|
||||||
config_path: /usr/local/etc/scriptorium/config.yml
|
config_path: /usr/local/etc/scriptorium/config.yml
|
||||||
@@ -130,12 +222,19 @@ scriptorium:
|
|||||||
timeout: 10m
|
timeout: 10m
|
||||||
inputs:
|
inputs:
|
||||||
transcript:
|
transcript:
|
||||||
source: narratio.transcript.trimmed
|
source: narratio.transcript.final_trimmed
|
||||||
required: true
|
required: true
|
||||||
previous_recap:
|
previous_recap:
|
||||||
source: previous_session_artifact
|
source: narratio.previous_session.artifact.session_recap
|
||||||
artifact: session_recap
|
required: false
|
||||||
path: ""
|
players:
|
||||||
|
source: narratio.input.players
|
||||||
|
required: true
|
||||||
|
party:
|
||||||
|
source: narratio.input.party
|
||||||
|
required: true
|
||||||
|
glossary:
|
||||||
|
source: narratio.input.glossary
|
||||||
required: false
|
required: false
|
||||||
vars:
|
vars:
|
||||||
session_id: true
|
session_id: true
|
||||||
@@ -160,23 +259,13 @@ scriptorium:
|
|||||||
source: narratio.artifact.session_recap
|
source: narratio.artifact.session_recap
|
||||||
required: true
|
required: true
|
||||||
transcript:
|
transcript:
|
||||||
source: narratio.transcript.trimmed
|
source: narratio.transcript.final_trimmed
|
||||||
required: true
|
required: true
|
||||||
vars:
|
vars:
|
||||||
session_id: true
|
session_id: true
|
||||||
campaign_name: true
|
campaign_name: true
|
||||||
output_kind: player_handout
|
output_kind: player_handout
|
||||||
|
|
||||||
analyzer:
|
|
||||||
# Optional adapter settings.
|
|
||||||
binary_path: ""
|
|
||||||
timeout: 2m
|
|
||||||
artifacts:
|
|
||||||
output_dir: ""
|
|
||||||
types: []
|
|
||||||
|
|
||||||
notification:
|
notification:
|
||||||
# Optional notification settings.
|
# No delivery provider is currently implemented.
|
||||||
backend: ""
|
mode: noop
|
||||||
recipient: ""
|
|
||||||
timeout: 30s
|
|
||||||
|
|||||||
@@ -1,2 +1,6 @@
|
|||||||
|
campaigns:
|
||||||
|
root: /usr/local/share/narratio/campaigns
|
||||||
|
default_campaign_id: sample-campaign
|
||||||
|
|
||||||
whisperx:
|
whisperx:
|
||||||
transcribe_url: "https://transcription.example.com/transcribe"
|
transcribe_url: "https://transcription.example.com/transcribe"
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
workspace:
|
workspace:
|
||||||
root: /var/lib/narratio/workspace
|
root: /var/lib/narratio/workspace
|
||||||
cleanup_after_archive: true
|
cleanup_after_publish: true
|
||||||
|
|
||||||
storage:
|
storage:
|
||||||
backend: s3
|
backend: s3
|
||||||
@@ -11,22 +11,32 @@ storage:
|
|||||||
access_key_id_env: OBJECT_STORAGE_KEY_ID
|
access_key_id_env: OBJECT_STORAGE_KEY_ID
|
||||||
secret_access_key_env: OBJECT_STORAGE_KEY
|
secret_access_key_env: OBJECT_STORAGE_KEY
|
||||||
|
|
||||||
|
campaigns:
|
||||||
|
root: /usr/local/share/narratio/campaigns
|
||||||
|
default_campaign_id: sample-campaign
|
||||||
|
|
||||||
spool:
|
spool:
|
||||||
root: /var/spool/narratio
|
root: /var/spool/narratio
|
||||||
delete_audio_after_archive: true
|
delete_audio_after_publish: true
|
||||||
|
|
||||||
archive:
|
publish:
|
||||||
enabled: true
|
enabled: true
|
||||||
upload_run: true
|
upload_run: true
|
||||||
promote_artifacts:
|
outputs:
|
||||||
- from: transcripts/trimmed.json
|
- source: narratio.transcript.final_trimmed
|
||||||
to: transcripts/trimmed.json
|
dest: transcripts/final.trimmed.json
|
||||||
required: true
|
required: true
|
||||||
- from: artifacts/session_recap.md
|
- source: narratio.transcript.final_markdown
|
||||||
to: artifacts/session_recap.md
|
dest: transcripts/final.md
|
||||||
required: true
|
required: true
|
||||||
- from: artifacts/player_handout.md
|
- source: narratio.transcript.final_trimmed_markdown
|
||||||
to: artifacts/player_handout.md
|
dest: transcripts/final.trimmed.md
|
||||||
|
required: true
|
||||||
|
- source: narratio.artifact.session_recap
|
||||||
|
dest: artifacts/session_recap.md
|
||||||
|
required: true
|
||||||
|
- source: narratio.artifact.player_handout
|
||||||
|
dest: artifacts/player_handout.md
|
||||||
required: false
|
required: false
|
||||||
|
|
||||||
whisperx:
|
whisperx:
|
||||||
@@ -57,13 +67,10 @@ audita:
|
|||||||
report: true
|
report: true
|
||||||
|
|
||||||
normalize:
|
normalize:
|
||||||
output_path: transcripts/normalized.json
|
output_path: transcripts/final.json
|
||||||
output_schema: seriatim-intermediate
|
output_schema: seriatim-intermediate
|
||||||
report: true
|
report: true
|
||||||
|
|
||||||
trim:
|
|
||||||
enabled: false
|
|
||||||
|
|
||||||
scriptorium:
|
scriptorium:
|
||||||
binary: scriptorium
|
binary: scriptorium
|
||||||
config_path: /usr/local/etc/scriptorium/config.yml
|
config_path: /usr/local/etc/scriptorium/config.yml
|
||||||
@@ -78,11 +85,19 @@ scriptorium:
|
|||||||
timeout: 10m
|
timeout: 10m
|
||||||
inputs:
|
inputs:
|
||||||
transcript:
|
transcript:
|
||||||
source: narratio.transcript.trimmed
|
source: narratio.transcript.final_trimmed
|
||||||
required: true
|
required: true
|
||||||
previous_recap:
|
previous_recap:
|
||||||
source: previous_session_artifact
|
source: narratio.previous_session.artifact.session_recap
|
||||||
artifact: session_recap
|
required: false
|
||||||
|
players:
|
||||||
|
source: narratio.input.players
|
||||||
|
required: true
|
||||||
|
party:
|
||||||
|
source: narratio.input.party
|
||||||
|
required: true
|
||||||
|
glossary:
|
||||||
|
source: narratio.input.glossary
|
||||||
required: false
|
required: false
|
||||||
vars:
|
vars:
|
||||||
session_id: true
|
session_id: true
|
||||||
@@ -103,14 +118,11 @@ scriptorium:
|
|||||||
source: narratio.artifact.session_recap
|
source: narratio.artifact.session_recap
|
||||||
required: true
|
required: true
|
||||||
transcript:
|
transcript:
|
||||||
source: narratio.transcript.trimmed
|
source: narratio.transcript.final_trimmed
|
||||||
required: true
|
required: true
|
||||||
vars:
|
vars:
|
||||||
session_id: true
|
session_id: true
|
||||||
output_kind: player_handout
|
output_kind: player_handout
|
||||||
|
|
||||||
analyzer:
|
|
||||||
timeout: 2m
|
|
||||||
|
|
||||||
notification:
|
notification:
|
||||||
timeout: 30s
|
mode: noop
|
||||||
|
|||||||
32
examples/production-testing/conf.d/artifacts.yml
Normal file
32
examples/production-testing/conf.d/artifacts.yml
Normal file
@@ -0,0 +1,32 @@
|
|||||||
|
scriptorium:
|
||||||
|
binary: scriptorium
|
||||||
|
config_path: ./scriptorium/config.yml
|
||||||
|
artifact_families:
|
||||||
|
character_meta:
|
||||||
|
enabled: true
|
||||||
|
for_each: party.characters
|
||||||
|
prompt_id: dnd.character_meta
|
||||||
|
output_path_pattern: artifacts/characters/{character_id}/meta.md
|
||||||
|
member_vars:
|
||||||
|
character_id: character_id
|
||||||
|
character_name: character.name
|
||||||
|
player_name: player.name
|
||||||
|
class_summary: character.class_summary
|
||||||
|
character_items:
|
||||||
|
enabled: true
|
||||||
|
for_each: party.characters
|
||||||
|
prompt_id: dnd.character_items
|
||||||
|
output_path_pattern: artifacts/characters/{character_id}/items.md
|
||||||
|
member_dependencies: [character_meta]
|
||||||
|
inputs:
|
||||||
|
character_meta:
|
||||||
|
source: narratio.member_artifact.character_meta
|
||||||
|
required: true
|
||||||
|
member_vars:
|
||||||
|
character_id: character_id
|
||||||
|
character_name: character.name
|
||||||
|
aliases: character.alias_summary
|
||||||
|
publish:
|
||||||
|
enabled: true
|
||||||
|
required: false
|
||||||
|
dest_pattern: artifacts/characters/{character_id}/items.md
|
||||||
12
examples/production-testing/conf.d/platform.yml
Normal file
12
examples/production-testing/conf.d/platform.yml
Normal file
@@ -0,0 +1,12 @@
|
|||||||
|
workspace:
|
||||||
|
root: ./workspace
|
||||||
|
|
||||||
|
campaigns:
|
||||||
|
root: ../../campaigns
|
||||||
|
default_campaign_id: sample-campaign
|
||||||
|
|
||||||
|
cache:
|
||||||
|
root: ./cache
|
||||||
|
|
||||||
|
spool:
|
||||||
|
root: ./spool
|
||||||
7
examples/production-testing/conf.d/publish.yml
Normal file
7
examples/production-testing/conf.d/publish.yml
Normal file
@@ -0,0 +1,7 @@
|
|||||||
|
publish:
|
||||||
|
enabled: true
|
||||||
|
upload_run: false
|
||||||
|
outputs:
|
||||||
|
- source: narratio.transcript.final_markdown
|
||||||
|
dest: transcripts/final.md
|
||||||
|
required: true
|
||||||
2
examples/production-testing/conf.d/storage.yml
Normal file
2
examples/production-testing/conf.d/storage.yml
Normal file
@@ -0,0 +1,2 @@
|
|||||||
|
storage:
|
||||||
|
backend: local
|
||||||
16
examples/production-testing/conf.d/transcript.yml
Normal file
16
examples/production-testing/conf.d/transcript.yml
Normal file
@@ -0,0 +1,16 @@
|
|||||||
|
whisperx:
|
||||||
|
transcribe_url: https://transcription.example.com/transcribe
|
||||||
|
language: en
|
||||||
|
|
||||||
|
seriatim:
|
||||||
|
binary: seriatim
|
||||||
|
output_schema: seriatim-intermediate
|
||||||
|
|
||||||
|
audita:
|
||||||
|
binary: audita
|
||||||
|
modules: [glossary, grammar]
|
||||||
|
output_schema: audita-v1
|
||||||
|
|
||||||
|
normalize:
|
||||||
|
output_path: transcripts/final.json
|
||||||
|
output_schema: seriatim-intermediate
|
||||||
15
examples/production-testing/pipeline.yml
Normal file
15
examples/production-testing/pipeline.yml
Normal file
@@ -0,0 +1,15 @@
|
|||||||
|
# Copyable production/testing pipeline entry point. Every fragment is named
|
||||||
|
# explicitly; Narratio never scans conf.d automatically.
|
||||||
|
composition:
|
||||||
|
imports:
|
||||||
|
- conf.d/platform.yml
|
||||||
|
- conf.d/storage.yml
|
||||||
|
- conf.d/transcript.yml
|
||||||
|
- conf.d/artifacts.yml
|
||||||
|
- conf.d/publish.yml
|
||||||
|
default_profile: production
|
||||||
|
profiles:
|
||||||
|
production:
|
||||||
|
overlay: profiles/production.yml
|
||||||
|
testing:
|
||||||
|
overlay: profiles/testing.yml
|
||||||
10
examples/production-testing/profiles/production.yml
Normal file
10
examples/production-testing/profiles/production.yml
Normal file
@@ -0,0 +1,10 @@
|
|||||||
|
audita:
|
||||||
|
model: narratio-production-model-placeholder
|
||||||
|
validation_model: narratio-production-validator-placeholder
|
||||||
|
|
||||||
|
scriptorium:
|
||||||
|
artifact_families:
|
||||||
|
character_meta:
|
||||||
|
profile_id: production-placeholder
|
||||||
|
character_items:
|
||||||
|
profile_id: production-placeholder
|
||||||
14
examples/production-testing/profiles/testing.yml
Normal file
14
examples/production-testing/profiles/testing.yml
Normal file
@@ -0,0 +1,14 @@
|
|||||||
|
audita:
|
||||||
|
model: narratio-testing-model-placeholder
|
||||||
|
validation_model: narratio-testing-validator-placeholder
|
||||||
|
|
||||||
|
scriptorium:
|
||||||
|
artifacts:
|
||||||
|
testing_notes:
|
||||||
|
enabled: false
|
||||||
|
output_path: artifacts/testing-notes.md
|
||||||
|
artifact_families:
|
||||||
|
character_meta:
|
||||||
|
profile_id: testing-placeholder
|
||||||
|
character_items:
|
||||||
|
profile_id: testing-placeholder
|
||||||
@@ -1,9 +1,5 @@
|
|||||||
session_id: 2026-05-03
|
session_id: 2026-05-03
|
||||||
campaign: sample-campaign
|
|
||||||
date: 2026-05-03
|
date: 2026-05-03
|
||||||
title: Sample Session
|
title: Sample Session
|
||||||
inputs:
|
inputs:
|
||||||
audio_dir: ./audio
|
audio_dir: ./audio
|
||||||
speakers_file: ./examples/speakers.yml
|
|
||||||
autocorrect_file: ./examples/autocorrect.yml
|
|
||||||
glossary_file: ./examples/glossary.yml
|
|
||||||
|
|||||||
@@ -1,10 +1,6 @@
|
|||||||
session_id: 2026-05-03
|
session_id: 2026-05-03
|
||||||
campaign: sample-campaign
|
|
||||||
date: 2026-05-03
|
date: 2026-05-03
|
||||||
title: Sample Session
|
title: Sample Session
|
||||||
inputs:
|
inputs:
|
||||||
audio_s3:
|
audio_s3:
|
||||||
prefix: audio/
|
prefix: audio/
|
||||||
speakers_file: ./examples/speakers.yml
|
|
||||||
autocorrect_file: ./examples/autocorrect.yml
|
|
||||||
glossary_file: ./examples/glossary.yml
|
|
||||||
|
|||||||
@@ -1,7 +1,3 @@
|
|||||||
session_id: "{{ session_id }}"
|
session_id: "{{ session_id }}"
|
||||||
campaign: sample-campaign
|
|
||||||
inputs:
|
inputs:
|
||||||
audio_dir: ./audio
|
audio_dir: ./audio
|
||||||
speakers_file: ./examples/speakers.yml
|
|
||||||
autocorrect_file: ./examples/autocorrect.yml
|
|
||||||
glossary_file: ./examples/glossary.yml
|
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
match:
|
match:
|
||||||
- speaker: "Eric Rakestraw"
|
- speaker: "Example Speaker"
|
||||||
match:
|
match:
|
||||||
- "Eric_Rakestraw"
|
- "Example_Speaker"
|
||||||
- "Eric"
|
- "Example"
|
||||||
|
|||||||
1
go.mod
1
go.mod
@@ -7,6 +7,7 @@ require (
|
|||||||
github.com/aws/aws-sdk-go-v2/credentials v1.19.16
|
github.com/aws/aws-sdk-go-v2/credentials v1.19.16
|
||||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.101.0
|
github.com/aws/aws-sdk-go-v2/service/s3 v1.101.0
|
||||||
github.com/aws/smithy-go v1.25.1
|
github.com/aws/smithy-go v1.25.1
|
||||||
|
golang.org/x/sys v0.47.0
|
||||||
gopkg.in/yaml.v3 v3.0.1
|
gopkg.in/yaml.v3 v3.0.1
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
2
go.sum
2
go.sum
@@ -34,6 +34,8 @@ github.com/aws/aws-sdk-go-v2/service/sts v1.42.1 h1:F/M5Y9I3nwr2IEpshZgh1GeHpOIt
|
|||||||
github.com/aws/aws-sdk-go-v2/service/sts v1.42.1/go.mod h1:mTNxImtovCOEEuD65mKW7DCsL+2gjEH+RPEAexAzAio=
|
github.com/aws/aws-sdk-go-v2/service/sts v1.42.1/go.mod h1:mTNxImtovCOEEuD65mKW7DCsL+2gjEH+RPEAexAzAio=
|
||||||
github.com/aws/smithy-go v1.25.1 h1:J8ERsGSU7d+aCmdQur5Txg6bVoYelvQJgtZehD12GkI=
|
github.com/aws/smithy-go v1.25.1 h1:J8ERsGSU7d+aCmdQur5Txg6bVoYelvQJgtZehD12GkI=
|
||||||
github.com/aws/smithy-go v1.25.1/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc=
|
github.com/aws/smithy-go v1.25.1/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc=
|
||||||
|
golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs=
|
||||||
|
golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||||
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM=
|
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM=
|
||||||
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
|
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
|
||||||
gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=
|
gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=
|
||||||
|
|||||||
@@ -1,40 +0,0 @@
|
|||||||
package analyzer
|
|
||||||
|
|
||||||
import "context"
|
|
||||||
|
|
||||||
// NoopRunner is a deterministic no-op analyzer adapter.
|
|
||||||
type NoopRunner struct{}
|
|
||||||
|
|
||||||
// Run returns the requested output path with placeholder metadata.
|
|
||||||
func (n *NoopRunner) Run(ctx context.Context, req AnalyzeRequest) (AnalyzeResult, error) {
|
|
||||||
if err := ctx.Err(); err != nil {
|
|
||||||
return AnalyzeResult{}, err
|
|
||||||
}
|
|
||||||
return AnalyzeResult{ArtifactPath: req.OutputPath, Metadata: map[string]any{"placeholder": true}}, nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// FakeRunner captures analyze requests and returns deterministic responses.
|
|
||||||
type FakeRunner struct {
|
|
||||||
Requests []AnalyzeRequest
|
|
||||||
Err error
|
|
||||||
Result AnalyzeResult
|
|
||||||
}
|
|
||||||
|
|
||||||
// Run records request and returns configured response.
|
|
||||||
func (f *FakeRunner) Run(ctx context.Context, req AnalyzeRequest) (AnalyzeResult, error) {
|
|
||||||
if err := ctx.Err(); err != nil {
|
|
||||||
return AnalyzeResult{}, err
|
|
||||||
}
|
|
||||||
f.Requests = append(f.Requests, req)
|
|
||||||
if f.Err != nil {
|
|
||||||
return AnalyzeResult{}, f.Err
|
|
||||||
}
|
|
||||||
res := f.Result
|
|
||||||
if res.ArtifactPath == "" {
|
|
||||||
res.ArtifactPath = req.OutputPath
|
|
||||||
}
|
|
||||||
if res.Metadata == nil {
|
|
||||||
res.Metadata = map[string]any{"fake": true}
|
|
||||||
}
|
|
||||||
return res, nil
|
|
||||||
}
|
|
||||||
@@ -1,31 +0,0 @@
|
|||||||
package analyzer
|
|
||||||
|
|
||||||
import (
|
|
||||||
"context"
|
|
||||||
"errors"
|
|
||||||
"testing"
|
|
||||||
)
|
|
||||||
|
|
||||||
func TestFakeRunnerCapturesRequestAndReturnsPath(t *testing.T) {
|
|
||||||
fake := &FakeRunner{}
|
|
||||||
req := AnalyzeRequest{ArtifactType: "session-log", OutputPath: "artifacts/session-log.md"}
|
|
||||||
|
|
||||||
res, err := fake.Run(context.Background(), req)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("Run() error = %v", err)
|
|
||||||
}
|
|
||||||
if len(fake.Requests) != 1 || fake.Requests[0].ArtifactType != "session-log" {
|
|
||||||
t.Fatalf("requests = %#v, want captured request", fake.Requests)
|
|
||||||
}
|
|
||||||
if res.ArtifactPath != req.OutputPath {
|
|
||||||
t.Fatalf("artifact path = %q, want %q", res.ArtifactPath, req.OutputPath)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestFakeRunnerError(t *testing.T) {
|
|
||||||
fake := &FakeRunner{Err: errors.New("boom")}
|
|
||||||
_, err := fake.Run(context.Background(), AnalyzeRequest{})
|
|
||||||
if err == nil {
|
|
||||||
t.Fatal("expected error, got nil")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,28 +0,0 @@
|
|||||||
// Package analyzer declares the adapter contract for artifact analysis generation.
|
|
||||||
package analyzer
|
|
||||||
|
|
||||||
import "context"
|
|
||||||
|
|
||||||
// TODO: implement analyzer integration once the analyzer contract is finalized.
|
|
||||||
|
|
||||||
// Runner is the adapter boundary for analyzer invocations.
|
|
||||||
type Runner interface {
|
|
||||||
Run(ctx context.Context, req AnalyzeRequest) (AnalyzeResult, error)
|
|
||||||
}
|
|
||||||
|
|
||||||
// AnalyzeRequest describes one analyzer artifact generation request.
|
|
||||||
type AnalyzeRequest struct {
|
|
||||||
ArtifactType string
|
|
||||||
ProcessedTranscriptPath string
|
|
||||||
ContextReferences []string
|
|
||||||
OutputPath string
|
|
||||||
GeneratedConfigPath string
|
|
||||||
StdoutLogPath string
|
|
||||||
StderrLogPath string
|
|
||||||
}
|
|
||||||
|
|
||||||
// AnalyzeResult describes analyzer output.
|
|
||||||
type AnalyzeResult struct {
|
|
||||||
ArtifactPath string
|
|
||||||
Metadata map[string]any
|
|
||||||
}
|
|
||||||
@@ -4,10 +4,10 @@ import (
|
|||||||
"context"
|
"context"
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
"fmt"
|
"fmt"
|
||||||
"os"
|
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||||
|
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||||
)
|
)
|
||||||
|
|
||||||
// NoopRunner is a deterministic no-op audita adapter.
|
// NoopRunner is a deterministic no-op audita adapter.
|
||||||
@@ -84,17 +84,17 @@ func materializePlaceholders(req PolishRequest) error {
|
|||||||
"merged_transcript_path": req.MergedTranscriptPath,
|
"merged_transcript_path": req.MergedTranscriptPath,
|
||||||
"output_path": req.OutputProcessedPath,
|
"output_path": req.OutputProcessedPath,
|
||||||
}
|
}
|
||||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if req.StdoutLogPath != "" {
|
if req.StdoutLogPath != "" {
|
||||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("audita noop/fake stdout placeholder\n"), 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("audita noop/fake stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if req.StderrLogPath != "" {
|
if req.StderrLogPath != "" {
|
||||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("audita noop/fake stderr placeholder\n"), 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("audita noop/fake stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -117,14 +117,14 @@ func writeJSONIfRequested(path string, payload any) error {
|
|||||||
if path == "" {
|
if path == "" {
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
if err := fileops.EnsureWorkspaceDirectory(filepath.Dir(path)); err != nil {
|
||||||
return fmt.Errorf("create parent directory %q: %w", filepath.Dir(path), err)
|
return fmt.Errorf("create parent directory %q: %w", filepath.Dir(path), err)
|
||||||
}
|
}
|
||||||
data, err := json.Marshal(payload)
|
data, err := json.Marshal(payload)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("marshal placeholder json for %q: %w", path, err)
|
return fmt.Errorf("marshal placeholder json for %q: %w", path, err)
|
||||||
}
|
}
|
||||||
if err := subprocess.WriteFileAtomic(path, data, 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(path, data, fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write placeholder json %q: %w", path, err)
|
return fmt.Errorf("write placeholder json %q: %w", path, err)
|
||||||
}
|
}
|
||||||
return nil
|
return nil
|
||||||
|
|||||||
@@ -14,7 +14,7 @@ func TestFakeRunnerCapturesRequestAndReturnsPath(t *testing.T) {
|
|||||||
dir := t.TempDir()
|
dir := t.TempDir()
|
||||||
req := PolishRequest{
|
req := PolishRequest{
|
||||||
GeneratedConfigPath: filepath.Join(dir, "config", "audita.yml"),
|
GeneratedConfigPath: filepath.Join(dir, "config", "audita.yml"),
|
||||||
OutputProcessedPath: filepath.Join(dir, "transcripts", "processed.json"),
|
OutputProcessedPath: filepath.Join(dir, "transcripts", "polished.json"),
|
||||||
StdoutLogPath: filepath.Join(dir, "logs", "audita.stdout.log"),
|
StdoutLogPath: filepath.Join(dir, "logs", "audita.stdout.log"),
|
||||||
StderrLogPath: filepath.Join(dir, "logs", "audita.stderr.log"),
|
StderrLogPath: filepath.Join(dir, "logs", "audita.stderr.log"),
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -6,8 +6,6 @@ import (
|
|||||||
"time"
|
"time"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TODO: implement a real Audita subprocess/service adapter.
|
|
||||||
|
|
||||||
// Runner is the adapter boundary for audita polish invocations.
|
// Runner is the adapter boundary for audita polish invocations.
|
||||||
type Runner interface {
|
type Runner interface {
|
||||||
Run(ctx context.Context, req PolishRequest) (PolishResult, error)
|
Run(ctx context.Context, req PolishRequest) (PolishResult, error)
|
||||||
@@ -15,25 +13,15 @@ type Runner interface {
|
|||||||
|
|
||||||
// PolishRequest describes an audita invocation.
|
// PolishRequest describes an audita invocation.
|
||||||
type PolishRequest struct {
|
type PolishRequest struct {
|
||||||
GeneratedConfigPath string
|
GeneratedConfigPath string
|
||||||
MergedTranscriptPath string
|
MergedTranscriptPath string
|
||||||
OutputProcessedPath string
|
OutputProcessedPath string
|
||||||
GlossaryPath string
|
GlossaryPath string
|
||||||
ReportPath string
|
ReportPath string
|
||||||
WorkDir string
|
WorkDir string
|
||||||
Modules []string
|
Modules []string
|
||||||
BaseURL string
|
StdoutLogPath string
|
||||||
Model string
|
StderrLogPath string
|
||||||
TranscriptDescription string
|
|
||||||
ConfigPath string
|
|
||||||
OutputSchema string
|
|
||||||
WorkDirRetention string
|
|
||||||
TotalLLMConcurrency *int
|
|
||||||
ProposalLLMConcurrency *int
|
|
||||||
ValidationModel string
|
|
||||||
ValidationLLMConcurrency *int
|
|
||||||
StdoutLogPath string
|
|
||||||
StderrLogPath string
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// PolishResult describes a polish output.
|
// PolishResult describes a polish output.
|
||||||
|
|||||||
@@ -11,8 +11,15 @@ import (
|
|||||||
"time"
|
"time"
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||||
|
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// MaxProcessedOutputBytes bounds Audita's processed-transcript JSON result.
|
||||||
|
const MaxProcessedOutputBytes int64 = 64 * 1024 * 1024
|
||||||
|
|
||||||
|
// MaxReportOutputBytes bounds Audita's optional report JSON result.
|
||||||
|
const MaxReportOutputBytes int64 = 16 * 1024 * 1024
|
||||||
|
|
||||||
// SubprocessRunnerConfig defines deterministic settings for Audita CLI execution.
|
// SubprocessRunnerConfig defines deterministic settings for Audita CLI execution.
|
||||||
type SubprocessRunnerConfig struct {
|
type SubprocessRunnerConfig struct {
|
||||||
Binary string
|
Binary string
|
||||||
@@ -207,12 +214,13 @@ func (r *SubprocessRunner) Run(ctx context.Context, req PolishRequest) (PolishRe
|
|||||||
}
|
}
|
||||||
|
|
||||||
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
|
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
|
||||||
Executable: r.binary,
|
Executable: r.binary,
|
||||||
Args: args,
|
Args: args,
|
||||||
Timeout: r.timeout,
|
Timeout: r.timeout,
|
||||||
EnvOverrides: env,
|
EnvOverrides: env,
|
||||||
StdoutLogPath: req.StdoutLogPath,
|
DiagnosticOwner: "audita",
|
||||||
StderrLogPath: req.StderrLogPath,
|
StdoutLogPath: req.StdoutLogPath,
|
||||||
|
StderrLogPath: req.StderrLogPath,
|
||||||
})
|
})
|
||||||
if err != nil {
|
if err != nil {
|
||||||
wrappedMessage := fmt.Sprintf(
|
wrappedMessage := fmt.Sprintf(
|
||||||
@@ -370,13 +378,13 @@ func (r *SubprocessRunner) writeInvocationConfig(req PolishRequest, args []strin
|
|||||||
"credential_env_var": r.llmAPIKeyEnv,
|
"credential_env_var": r.llmAPIKeyEnv,
|
||||||
"credential_present": credentialPresent,
|
"credential_present": credentialPresent,
|
||||||
}
|
}
|
||||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
|
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
|
||||||
}
|
}
|
||||||
|
|
||||||
func validateProcessedOutput(path string) error {
|
func validateProcessedOutput(path string) error {
|
||||||
data, err := os.ReadFile(path)
|
data, err := readAuditaResult(path, MaxProcessedOutputBytes, "processed transcript")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("read file: %w", err)
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
var payload map[string]any
|
var payload map[string]any
|
||||||
@@ -405,9 +413,9 @@ func addSubprocessStreamHint(message string, runErr error) string {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func validateJSONFile(path string) error {
|
func validateJSONFile(path string) error {
|
||||||
data, err := os.ReadFile(path)
|
data, err := readAuditaResult(path, MaxReportOutputBytes, "report")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("read file: %w", err)
|
return err
|
||||||
}
|
}
|
||||||
var v any
|
var v any
|
||||||
if err := json.Unmarshal(data, &v); err != nil {
|
if err := json.Unmarshal(data, &v); err != nil {
|
||||||
@@ -415,3 +423,11 @@ func validateJSONFile(path string) error {
|
|||||||
}
|
}
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func readAuditaResult(path string, limit int64, category string) ([]byte, error) {
|
||||||
|
data, err := fileops.ReadRegularFile(path, limit)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("audita %s result exceeds or cannot be read within %d-byte limit: %w", category, limit, err)
|
||||||
|
}
|
||||||
|
return data, nil
|
||||||
|
}
|
||||||
|
|||||||
@@ -52,9 +52,9 @@ func TestSubprocessRunnerSuccessArgsEnvAndValidation(t *testing.T) {
|
|||||||
dir := t.TempDir()
|
dir := t.TempDir()
|
||||||
req := PolishRequest{
|
req := PolishRequest{
|
||||||
GeneratedConfigPath: filepath.Join(dir, "audita.generated.yml"),
|
GeneratedConfigPath: filepath.Join(dir, "audita.generated.yml"),
|
||||||
MergedTranscriptPath: filepath.Join(dir, "merged.json"),
|
MergedTranscriptPath: filepath.Join(dir, "base.json"),
|
||||||
GlossaryPath: filepath.Join(dir, "glossary.yml"),
|
GlossaryPath: filepath.Join(dir, "glossary.yml"),
|
||||||
OutputProcessedPath: filepath.Join(dir, "processed.json"),
|
OutputProcessedPath: filepath.Join(dir, "polished.json"),
|
||||||
ReportPath: filepath.Join(dir, "audita.report.json"),
|
ReportPath: filepath.Join(dir, "audita.report.json"),
|
||||||
WorkDir: filepath.Join(dir, "artifacts", "audita-work"),
|
WorkDir: filepath.Join(dir, "artifacts", "audita-work"),
|
||||||
StdoutLogPath: filepath.Join(dir, "audita.stdout.log"),
|
StdoutLogPath: filepath.Join(dir, "audita.stdout.log"),
|
||||||
@@ -189,7 +189,7 @@ func TestSubprocessRunnerUnconfiguredCredentialEnvOmitsCredential(t *testing.T)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestSubprocessRunnerInheritsParentEnvironment(t *testing.T) {
|
func TestSubprocessRunnerOmitsUnspecifiedParentEnvironment(t *testing.T) {
|
||||||
if runtime.GOOS == "windows" {
|
if runtime.GOOS == "windows" {
|
||||||
t.Skip("helper wrapper script uses /bin/sh")
|
t.Skip("helper wrapper script uses /bin/sh")
|
||||||
}
|
}
|
||||||
@@ -214,8 +214,8 @@ func TestSubprocessRunnerInheritsParentEnvironment(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
rec := readAuditaHelperRecord(t, recordPath)
|
rec := readAuditaHelperRecord(t, recordPath)
|
||||||
if rec.Env["AUDITA_INHERITED_MARKER"] != "inherited-from-parent" {
|
if rec.Env["AUDITA_INHERITED_MARKER"] != "" {
|
||||||
t.Fatalf("AUDITA_INHERITED_MARKER = %q, want inherited-from-parent", rec.Env["AUDITA_INHERITED_MARKER"])
|
t.Fatalf("AUDITA_INHERITED_MARKER = %q, want omitted from the child environment", rec.Env["AUDITA_INHERITED_MARKER"])
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -571,7 +571,7 @@ func mustAuditaRunner(t *testing.T, cfg SubprocessRunnerConfig) *SubprocessRunne
|
|||||||
func auditaReqForTest(t *testing.T, withReport bool) PolishRequest {
|
func auditaReqForTest(t *testing.T, withReport bool) PolishRequest {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
dir := t.TempDir()
|
dir := t.TempDir()
|
||||||
merged := filepath.Join(dir, "merged.json")
|
merged := filepath.Join(dir, "base.json")
|
||||||
glossary := filepath.Join(dir, "glossary.yml")
|
glossary := filepath.Join(dir, "glossary.yml")
|
||||||
writeAuditaTestFile(t, merged, `{"segments":[]}`)
|
writeAuditaTestFile(t, merged, `{"segments":[]}`)
|
||||||
writeAuditaTestFile(t, glossary, "terms: []\n")
|
writeAuditaTestFile(t, glossary, "terms: []\n")
|
||||||
@@ -579,7 +579,7 @@ func auditaReqForTest(t *testing.T, withReport bool) PolishRequest {
|
|||||||
GeneratedConfigPath: filepath.Join(dir, "audita.generated.yml"),
|
GeneratedConfigPath: filepath.Join(dir, "audita.generated.yml"),
|
||||||
MergedTranscriptPath: merged,
|
MergedTranscriptPath: merged,
|
||||||
GlossaryPath: glossary,
|
GlossaryPath: glossary,
|
||||||
OutputProcessedPath: filepath.Join(dir, "processed.json"),
|
OutputProcessedPath: filepath.Join(dir, "polished.json"),
|
||||||
WorkDir: filepath.Join(dir, "artifacts", "audita-work"),
|
WorkDir: filepath.Join(dir, "artifacts", "audita-work"),
|
||||||
StdoutLogPath: filepath.Join(dir, "audita.stdout.log"),
|
StdoutLogPath: filepath.Join(dir, "audita.stdout.log"),
|
||||||
StderrLogPath: filepath.Join(dir, "audita.stderr.log"),
|
StderrLogPath: filepath.Join(dir, "audita.stderr.log"),
|
||||||
|
|||||||
24
internal/adapters/notarius/fake.go
Normal file
24
internal/adapters/notarius/fake.go
Normal file
@@ -0,0 +1,24 @@
|
|||||||
|
package notarius
|
||||||
|
|
||||||
|
import "context"
|
||||||
|
|
||||||
|
// FakeRunner is a configurable in-memory runner for stage tests.
|
||||||
|
type FakeRunner struct {
|
||||||
|
Requests []RunRequest
|
||||||
|
Result RunResult
|
||||||
|
Err error
|
||||||
|
}
|
||||||
|
|
||||||
|
// Run records the request and returns the configured result or error.
|
||||||
|
func (f *FakeRunner) Run(ctx context.Context, req RunRequest) (RunResult, error) {
|
||||||
|
if err := ctx.Err(); err != nil {
|
||||||
|
return RunResult{}, err
|
||||||
|
}
|
||||||
|
copyRequest := req
|
||||||
|
copyRequest.References = append([]ReferenceBinding(nil), req.References...)
|
||||||
|
f.Requests = append(f.Requests, copyRequest)
|
||||||
|
if f.Err != nil {
|
||||||
|
return RunResult{}, f.Err
|
||||||
|
}
|
||||||
|
return f.Result, nil
|
||||||
|
}
|
||||||
160
internal/adapters/notarius/runner.go
Normal file
160
internal/adapters/notarius/runner.go
Normal file
@@ -0,0 +1,160 @@
|
|||||||
|
// Package notarius declares the adapter contract for Notarius CLI invocations.
|
||||||
|
package notarius
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
const ReceiptSchemaVersion = "notarius.run-result.v2"
|
||||||
|
|
||||||
|
// Runner is the adapter boundary for a complete Notarius pipeline invocation.
|
||||||
|
type Runner interface {
|
||||||
|
Run(ctx context.Context, req RunRequest) (RunResult, error)
|
||||||
|
}
|
||||||
|
|
||||||
|
// ReferenceBinding maps one normalized Notarius selector to an absolute
|
||||||
|
// external reference path.
|
||||||
|
type ReferenceBinding struct {
|
||||||
|
Selector string
|
||||||
|
Path string
|
||||||
|
}
|
||||||
|
|
||||||
|
// RunRequest contains the resolved inputs and diagnostic destinations for one invocation.
|
||||||
|
type RunRequest struct {
|
||||||
|
Binary string
|
||||||
|
ConfigPath string
|
||||||
|
PipelineID string
|
||||||
|
InputPath string
|
||||||
|
OutputRoot string
|
||||||
|
WorkingDirectory string
|
||||||
|
ReceiptPath string
|
||||||
|
LogPath string
|
||||||
|
Timeout time.Duration
|
||||||
|
References []ReferenceBinding
|
||||||
|
}
|
||||||
|
|
||||||
|
// Receipt is the transport-neutral successful run receipt.
|
||||||
|
type Receipt struct {
|
||||||
|
SchemaVersion string
|
||||||
|
RunID string
|
||||||
|
PipelineID string
|
||||||
|
OutputDirectory string
|
||||||
|
IndexFile string
|
||||||
|
NormalizedOutputCount int
|
||||||
|
RejectedOutputCount int
|
||||||
|
WarningGroupCount int
|
||||||
|
WarningOccurrenceCount int
|
||||||
|
DiagnosticGroupCount int
|
||||||
|
DiagnosticOccurrenceCount int
|
||||||
|
DiagnosticsTruncated bool
|
||||||
|
ValidationStatus string
|
||||||
|
ValidationSummaries []ValidationSummary
|
||||||
|
DebugDirectory string
|
||||||
|
}
|
||||||
|
|
||||||
|
// ValidationSummary retains the bounded outcome of one Notarius producer result.
|
||||||
|
type ValidationSummary struct {
|
||||||
|
Stage string
|
||||||
|
StepID string
|
||||||
|
LaneID string
|
||||||
|
ModuleKey string
|
||||||
|
ChunkID string
|
||||||
|
Status string
|
||||||
|
RejectingValidators []string
|
||||||
|
ReasonCodes []string
|
||||||
|
IncompleteValidators []string
|
||||||
|
ProducerAttemptCount int
|
||||||
|
TerminalAction string
|
||||||
|
}
|
||||||
|
|
||||||
|
// LaneDescriptor identifies one normalized lane payload discovered through the index.
|
||||||
|
type LaneDescriptor struct {
|
||||||
|
LaneID string
|
||||||
|
File string
|
||||||
|
Path string
|
||||||
|
MediaType string
|
||||||
|
ModuleKey string
|
||||||
|
SchemaID string
|
||||||
|
SchemaName string
|
||||||
|
SchemaVersion string
|
||||||
|
}
|
||||||
|
|
||||||
|
// PipelineDescriptor identifies a pipeline-wide artifact discovered through the index.
|
||||||
|
type PipelineDescriptor struct {
|
||||||
|
ArtifactKind string
|
||||||
|
File string
|
||||||
|
Path string
|
||||||
|
MediaType string
|
||||||
|
SchemaID string
|
||||||
|
SchemaName string
|
||||||
|
SchemaVersion string
|
||||||
|
}
|
||||||
|
|
||||||
|
// Index describes the validated bundle-management and artifact paths.
|
||||||
|
type Index struct {
|
||||||
|
Path string
|
||||||
|
ManifestFile string
|
||||||
|
ManifestPath string
|
||||||
|
RejectedFile string
|
||||||
|
RejectedPath string
|
||||||
|
WarningsFile string
|
||||||
|
WarningsPath string
|
||||||
|
DiagnosticsFile string
|
||||||
|
DiagnosticsPath string
|
||||||
|
Lanes []LaneDescriptor
|
||||||
|
ChunkMap *PipelineDescriptor
|
||||||
|
EvidenceContext *PipelineDescriptor
|
||||||
|
}
|
||||||
|
|
||||||
|
// RejectionSummary retains structured rejection identity without free-form messages.
|
||||||
|
type RejectionSummary struct {
|
||||||
|
Stage string
|
||||||
|
StepID string
|
||||||
|
LaneID string
|
||||||
|
ModuleKey string
|
||||||
|
ChunkID string
|
||||||
|
ValidatorName string
|
||||||
|
ReasonCode string
|
||||||
|
}
|
||||||
|
|
||||||
|
// WarningSummary retains structured warning identity without free-form messages.
|
||||||
|
type WarningSummary struct {
|
||||||
|
Disposition string
|
||||||
|
Category string
|
||||||
|
ReasonCode string
|
||||||
|
Origin DiagnosticOrigin
|
||||||
|
OccurrenceCount int
|
||||||
|
}
|
||||||
|
|
||||||
|
// DiagnosticOrigin identifies the framework-owned pipeline location of a finding.
|
||||||
|
type DiagnosticOrigin struct {
|
||||||
|
Stage string
|
||||||
|
StepID string
|
||||||
|
LaneID string
|
||||||
|
ModuleKey string
|
||||||
|
ValidatorKey string
|
||||||
|
}
|
||||||
|
|
||||||
|
// DiagnosticSummary retains bounded advisory or observation group metadata.
|
||||||
|
type DiagnosticSummary struct {
|
||||||
|
Disposition string
|
||||||
|
Category string
|
||||||
|
ReasonCode string
|
||||||
|
Origin DiagnosticOrigin
|
||||||
|
OccurrenceCount int
|
||||||
|
}
|
||||||
|
|
||||||
|
// RunResult describes a successfully decoded and validated Notarius bundle.
|
||||||
|
type RunResult struct {
|
||||||
|
Receipt Receipt
|
||||||
|
Index Index
|
||||||
|
BundleRoot string
|
||||||
|
ReceiptPath string
|
||||||
|
LogPath string
|
||||||
|
ExitCode int
|
||||||
|
Duration time.Duration
|
||||||
|
Rejections []RejectionSummary
|
||||||
|
Warnings []WarningSummary
|
||||||
|
Diagnostics []DiagnosticSummary
|
||||||
|
}
|
||||||
784
internal/adapters/notarius/subprocess.go
Normal file
784
internal/adapters/notarius/subprocess.go
Normal file
@@ -0,0 +1,784 @@
|
|||||||
|
package notarius
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||||
|
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||||
|
"gitea.maximumdirect.net/eric/narratio/internal/notariusref"
|
||||||
|
"gitea.maximumdirect.net/eric/narratio/internal/pathsafe"
|
||||||
|
)
|
||||||
|
|
||||||
|
const (
|
||||||
|
maxReceiptBytes = 1 << 20
|
||||||
|
maxIndexBytes = 4 << 20
|
||||||
|
maxSummaryBytes = 4 << 20
|
||||||
|
canonicalIndexFile = "index.json"
|
||||||
|
canonicalManifestFile = "manifest.json"
|
||||||
|
canonicalRejectedFile = "rejected.json"
|
||||||
|
canonicalWarningsFile = "warnings.json"
|
||||||
|
canonicalDiagnosticsFile = "diagnostics.json"
|
||||||
|
warningsSchemaVersion = "notarius.warnings.v2"
|
||||||
|
diagnosticsSchemaVersion = "notarius.diagnostics.v1"
|
||||||
|
maxWarningGroups = 128
|
||||||
|
maxDiagnosticGroups = 256
|
||||||
|
maxFindingSamples = 3
|
||||||
|
)
|
||||||
|
|
||||||
|
type subprocessRun func(context.Context, subprocess.RunRequest) (subprocess.RunResult, error)
|
||||||
|
|
||||||
|
// SubprocessRunner invokes Notarius through its public CLI.
|
||||||
|
type SubprocessRunner struct {
|
||||||
|
run subprocessRun
|
||||||
|
}
|
||||||
|
|
||||||
|
// NewSubprocessRunner constructs a production Notarius subprocess runner.
|
||||||
|
func NewSubprocessRunner() *SubprocessRunner {
|
||||||
|
return &SubprocessRunner{run: subprocess.Run}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Run executes a complete Notarius pipeline and discovers its published bundle.
|
||||||
|
func (r *SubprocessRunner) Run(ctx context.Context, req RunRequest) (RunResult, error) {
|
||||||
|
if r == nil || r.run == nil {
|
||||||
|
return RunResult{}, fmt.Errorf("notarius subprocess runner is nil")
|
||||||
|
}
|
||||||
|
references, err := validateRunRequest(req)
|
||||||
|
if err != nil {
|
||||||
|
return RunResult{}, err
|
||||||
|
}
|
||||||
|
|
||||||
|
args := []string{
|
||||||
|
"run", req.PipelineID,
|
||||||
|
"--config", req.ConfigPath,
|
||||||
|
"--input", req.InputPath,
|
||||||
|
"--output-dir", req.OutputRoot,
|
||||||
|
}
|
||||||
|
for _, reference := range references {
|
||||||
|
args = append(args, "--reference", reference.Selector+"="+reference.Path)
|
||||||
|
}
|
||||||
|
args = append(args, "--json")
|
||||||
|
processResult, err := r.run(ctx, subprocess.RunRequest{
|
||||||
|
Executable: req.Binary,
|
||||||
|
Args: args,
|
||||||
|
WorkingDir: req.WorkingDirectory,
|
||||||
|
Timeout: req.Timeout,
|
||||||
|
DiagnosticOwner: "notarius",
|
||||||
|
StdoutLogPath: req.ReceiptPath,
|
||||||
|
StderrLogPath: req.LogPath,
|
||||||
|
})
|
||||||
|
baseResult := RunResult{
|
||||||
|
ReceiptPath: req.ReceiptPath,
|
||||||
|
LogPath: req.LogPath,
|
||||||
|
ExitCode: processResult.ExitCode,
|
||||||
|
Duration: processResult.Duration,
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
return baseResult, fmt.Errorf("run notarius pipeline %q: %w", req.PipelineID, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
receipt, err := loadReceipt(req.ReceiptPath, req.PipelineID)
|
||||||
|
if err != nil {
|
||||||
|
return baseResult, err
|
||||||
|
}
|
||||||
|
bundleRoot, err := validateBundleRoot(req.OutputRoot, receipt.OutputDirectory)
|
||||||
|
if err != nil {
|
||||||
|
return baseResult, err
|
||||||
|
}
|
||||||
|
indexPath, err := resolveRegularFile(bundleRoot, receipt.IndexFile)
|
||||||
|
if err != nil {
|
||||||
|
return baseResult, fmt.Errorf("resolve receipt index file: %w", err)
|
||||||
|
}
|
||||||
|
index, err := loadIndex(bundleRoot, indexPath)
|
||||||
|
if err != nil {
|
||||||
|
return baseResult, err
|
||||||
|
}
|
||||||
|
rejections, err := loadRejections(index.RejectedPath)
|
||||||
|
if err != nil {
|
||||||
|
return baseResult, err
|
||||||
|
}
|
||||||
|
warnings, err := loadWarnings(index.WarningsPath)
|
||||||
|
if err != nil {
|
||||||
|
return baseResult, err
|
||||||
|
}
|
||||||
|
diagnostics, diagnosticOccurrences, diagnosticsTruncated, err := loadDiagnostics(index.DiagnosticsPath)
|
||||||
|
if err != nil {
|
||||||
|
return baseResult, err
|
||||||
|
}
|
||||||
|
if receipt.NormalizedOutputCount != len(index.Lanes) || receipt.RejectedOutputCount != len(rejections) ||
|
||||||
|
receipt.WarningGroupCount != len(warnings) || receipt.DiagnosticGroupCount != len(diagnostics) {
|
||||||
|
return baseResult, fmt.Errorf("notarius receipt counts do not match published bundle")
|
||||||
|
}
|
||||||
|
warningOccurrences, err := sumWarningOccurrences(warnings)
|
||||||
|
if err != nil {
|
||||||
|
return baseResult, err
|
||||||
|
}
|
||||||
|
if receipt.WarningOccurrenceCount != warningOccurrences ||
|
||||||
|
receipt.DiagnosticOccurrenceCount != diagnosticOccurrences ||
|
||||||
|
receipt.DiagnosticsTruncated != diagnosticsTruncated {
|
||||||
|
return baseResult, fmt.Errorf("notarius receipt occurrence counts do not match published bundle")
|
||||||
|
}
|
||||||
|
|
||||||
|
baseResult.Receipt = receipt
|
||||||
|
baseResult.Index = index
|
||||||
|
baseResult.BundleRoot = bundleRoot
|
||||||
|
baseResult.Rejections = rejections
|
||||||
|
baseResult.Warnings = warnings
|
||||||
|
baseResult.Diagnostics = diagnostics
|
||||||
|
return baseResult, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func validateRunRequest(req RunRequest) ([]ReferenceBinding, error) {
|
||||||
|
if strings.TrimSpace(req.Binary) == "" {
|
||||||
|
return nil, fmt.Errorf("notarius binary is required")
|
||||||
|
}
|
||||||
|
if strings.TrimSpace(req.PipelineID) == "" {
|
||||||
|
return nil, fmt.Errorf("notarius pipeline id is required")
|
||||||
|
}
|
||||||
|
if req.Timeout <= 0 {
|
||||||
|
return nil, fmt.Errorf("notarius timeout must be positive")
|
||||||
|
}
|
||||||
|
for label, path := range map[string]string{
|
||||||
|
"config": req.ConfigPath,
|
||||||
|
"input": req.InputPath,
|
||||||
|
"output root": req.OutputRoot,
|
||||||
|
"working directory": req.WorkingDirectory,
|
||||||
|
"receipt": req.ReceiptPath,
|
||||||
|
"log": req.LogPath,
|
||||||
|
} {
|
||||||
|
if strings.TrimSpace(path) == "" {
|
||||||
|
return nil, fmt.Errorf("notarius %s path is required", label)
|
||||||
|
}
|
||||||
|
if !filepath.IsAbs(path) {
|
||||||
|
return nil, fmt.Errorf("notarius %s path must be absolute", label)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if filepath.Clean(req.ReceiptPath) == filepath.Clean(req.LogPath) {
|
||||||
|
return nil, fmt.Errorf("notarius receipt and log paths must be different")
|
||||||
|
}
|
||||||
|
references := make([]ReferenceBinding, 0, len(req.References))
|
||||||
|
selectors := make(map[string]struct{}, len(req.References))
|
||||||
|
for index, binding := range req.References {
|
||||||
|
selector, err := notariusref.NormalizeSelector(binding.Selector)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("notarius reference %d selector: %w", index, err)
|
||||||
|
}
|
||||||
|
if _, duplicate := selectors[selector]; duplicate {
|
||||||
|
return nil, fmt.Errorf("notarius reference selector %q is duplicated", selector)
|
||||||
|
}
|
||||||
|
selectors[selector] = struct{}{}
|
||||||
|
if strings.TrimSpace(binding.Path) == "" {
|
||||||
|
return nil, fmt.Errorf("notarius reference %q path is required", selector)
|
||||||
|
}
|
||||||
|
if !filepath.IsAbs(binding.Path) {
|
||||||
|
return nil, fmt.Errorf("notarius reference %q path must be absolute", selector)
|
||||||
|
}
|
||||||
|
references = append(references, ReferenceBinding{Selector: selector, Path: binding.Path})
|
||||||
|
}
|
||||||
|
if err := requireRegularFile(req.ConfigPath); err != nil {
|
||||||
|
return nil, fmt.Errorf("validate notarius config path: %w", err)
|
||||||
|
}
|
||||||
|
if err := requireRegularFile(req.InputPath); err != nil {
|
||||||
|
return nil, fmt.Errorf("validate notarius input path: %w", err)
|
||||||
|
}
|
||||||
|
if err := requireDirectory(req.OutputRoot); err != nil {
|
||||||
|
return nil, fmt.Errorf("validate notarius output root: %w", err)
|
||||||
|
}
|
||||||
|
if err := requireDirectory(req.WorkingDirectory); err != nil {
|
||||||
|
return nil, fmt.Errorf("validate notarius working directory: %w", err)
|
||||||
|
}
|
||||||
|
if err := validateLogDestination(req.ReceiptPath); err != nil {
|
||||||
|
return nil, fmt.Errorf("validate notarius receipt path: %w", err)
|
||||||
|
}
|
||||||
|
if err := validateLogDestination(req.LogPath); err != nil {
|
||||||
|
return nil, fmt.Errorf("validate notarius log path: %w", err)
|
||||||
|
}
|
||||||
|
return references, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type receiptDocument struct {
|
||||||
|
SchemaVersion string `json:"schema_version"`
|
||||||
|
RunID string `json:"run_id"`
|
||||||
|
PipelineID string `json:"pipeline_id"`
|
||||||
|
OutputDirectory string `json:"output_directory"`
|
||||||
|
IndexFile string `json:"index_file"`
|
||||||
|
NormalizedOutputCount *int `json:"normalized_output_count"`
|
||||||
|
RejectedOutputCount *int `json:"rejected_output_count"`
|
||||||
|
WarningGroupCount *int `json:"warning_group_count"`
|
||||||
|
WarningOccurrenceCount *int `json:"warning_occurrence_count"`
|
||||||
|
DiagnosticGroupCount *int `json:"diagnostic_group_count"`
|
||||||
|
DiagnosticOccurrenceCount *int `json:"diagnostic_occurrence_count"`
|
||||||
|
DiagnosticsTruncated *bool `json:"diagnostics_truncated"`
|
||||||
|
ValidationStatus string `json:"validation_status"`
|
||||||
|
ValidationSummaries []validationSummaryDocument `json:"validation_summaries"`
|
||||||
|
DebugDirectory string `json:"debug_directory"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type validationSummaryDocument struct {
|
||||||
|
Stage string `json:"stage"`
|
||||||
|
StepID string `json:"step_id"`
|
||||||
|
LaneID string `json:"lane_id"`
|
||||||
|
ModuleKey string `json:"module_key"`
|
||||||
|
ChunkID string `json:"chunk_id"`
|
||||||
|
Status string `json:"status"`
|
||||||
|
RejectingValidators []string `json:"rejecting_validators"`
|
||||||
|
ReasonCodes []string `json:"reason_codes"`
|
||||||
|
IncompleteValidators []string `json:"incomplete_validators"`
|
||||||
|
ProducerAttemptCount *int `json:"producer_attempt_count"`
|
||||||
|
TerminalAction string `json:"terminal_action"`
|
||||||
|
}
|
||||||
|
|
||||||
|
func validValidationStatus(value string) bool {
|
||||||
|
switch value {
|
||||||
|
case "approved", "rejected", "incomplete":
|
||||||
|
return true
|
||||||
|
default:
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func validateValidationSummaries(documents []validationSummaryDocument) ([]ValidationSummary, error) {
|
||||||
|
summaries := make([]ValidationSummary, 0, len(documents))
|
||||||
|
for _, document := range documents {
|
||||||
|
if document.Status != "complete" && document.Status != "rejected" && document.Status != "incomplete" {
|
||||||
|
return nil, fmt.Errorf("notarius validation summary status %q is invalid", document.Status)
|
||||||
|
}
|
||||||
|
if document.ProducerAttemptCount == nil || *document.ProducerAttemptCount <= 0 || !validTerminalAction(document.TerminalAction) {
|
||||||
|
return nil, fmt.Errorf("notarius validation summary is missing required fields")
|
||||||
|
}
|
||||||
|
summaries = append(summaries, ValidationSummary{
|
||||||
|
Stage: document.Stage, StepID: document.StepID, LaneID: document.LaneID,
|
||||||
|
ModuleKey: document.ModuleKey, ChunkID: document.ChunkID, Status: document.Status,
|
||||||
|
RejectingValidators: append([]string(nil), document.RejectingValidators...),
|
||||||
|
ReasonCodes: append([]string(nil), document.ReasonCodes...),
|
||||||
|
IncompleteValidators: append([]string(nil), document.IncompleteValidators...),
|
||||||
|
ProducerAttemptCount: *document.ProducerAttemptCount, TerminalAction: document.TerminalAction,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
return summaries, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func validTerminalAction(value string) bool {
|
||||||
|
switch value {
|
||||||
|
case "accepted", "reject_output", "warn_continue", "fail_run":
|
||||||
|
return true
|
||||||
|
default:
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func loadReceipt(path, pipelineID string) (Receipt, error) {
|
||||||
|
var document receiptDocument
|
||||||
|
if err := decodeBoundedJSON(path, maxReceiptBytes, &document); err != nil {
|
||||||
|
return Receipt{}, fmt.Errorf("decode notarius receipt: %w", err)
|
||||||
|
}
|
||||||
|
if document.SchemaVersion != ReceiptSchemaVersion {
|
||||||
|
return Receipt{}, fmt.Errorf("unsupported notarius receipt schema version %q", document.SchemaVersion)
|
||||||
|
}
|
||||||
|
if strings.TrimSpace(document.RunID) == "" || strings.TrimSpace(document.PipelineID) == "" ||
|
||||||
|
strings.TrimSpace(document.OutputDirectory) == "" || strings.TrimSpace(document.ValidationStatus) == "" ||
|
||||||
|
document.NormalizedOutputCount == nil ||
|
||||||
|
document.RejectedOutputCount == nil || document.WarningGroupCount == nil ||
|
||||||
|
document.WarningOccurrenceCount == nil || document.DiagnosticGroupCount == nil ||
|
||||||
|
document.DiagnosticOccurrenceCount == nil || document.DiagnosticsTruncated == nil {
|
||||||
|
return Receipt{}, fmt.Errorf("notarius receipt is missing required fields")
|
||||||
|
}
|
||||||
|
if document.IndexFile != canonicalIndexFile {
|
||||||
|
return Receipt{}, fmt.Errorf("notarius receipt index_file %q is incompatible; want %q", document.IndexFile, canonicalIndexFile)
|
||||||
|
}
|
||||||
|
if document.PipelineID != pipelineID {
|
||||||
|
return Receipt{}, fmt.Errorf("notarius receipt pipeline id %q does not match requested pipeline %q", document.PipelineID, pipelineID)
|
||||||
|
}
|
||||||
|
if *document.NormalizedOutputCount < 0 || *document.RejectedOutputCount < 0 ||
|
||||||
|
*document.WarningGroupCount < 0 || *document.WarningOccurrenceCount < 0 ||
|
||||||
|
*document.DiagnosticGroupCount < 0 || *document.DiagnosticOccurrenceCount < 0 {
|
||||||
|
return Receipt{}, fmt.Errorf("notarius receipt counts must be non-negative")
|
||||||
|
}
|
||||||
|
if !validValidationStatus(document.ValidationStatus) {
|
||||||
|
return Receipt{}, fmt.Errorf("notarius receipt validation_status %q is invalid", document.ValidationStatus)
|
||||||
|
}
|
||||||
|
validationSummaries, err := validateValidationSummaries(document.ValidationSummaries)
|
||||||
|
if err != nil {
|
||||||
|
return Receipt{}, err
|
||||||
|
}
|
||||||
|
if !filepath.IsAbs(document.OutputDirectory) {
|
||||||
|
return Receipt{}, fmt.Errorf("notarius receipt output directory must be absolute")
|
||||||
|
}
|
||||||
|
if document.DebugDirectory != "" && !filepath.IsAbs(document.DebugDirectory) {
|
||||||
|
return Receipt{}, fmt.Errorf("notarius receipt debug directory must be absolute when present")
|
||||||
|
}
|
||||||
|
return Receipt{
|
||||||
|
SchemaVersion: document.SchemaVersion,
|
||||||
|
RunID: document.RunID,
|
||||||
|
PipelineID: document.PipelineID,
|
||||||
|
OutputDirectory: filepath.Clean(document.OutputDirectory),
|
||||||
|
IndexFile: document.IndexFile,
|
||||||
|
NormalizedOutputCount: *document.NormalizedOutputCount,
|
||||||
|
RejectedOutputCount: *document.RejectedOutputCount,
|
||||||
|
WarningGroupCount: *document.WarningGroupCount,
|
||||||
|
WarningOccurrenceCount: *document.WarningOccurrenceCount,
|
||||||
|
DiagnosticGroupCount: *document.DiagnosticGroupCount,
|
||||||
|
DiagnosticOccurrenceCount: *document.DiagnosticOccurrenceCount,
|
||||||
|
DiagnosticsTruncated: *document.DiagnosticsTruncated,
|
||||||
|
ValidationStatus: document.ValidationStatus,
|
||||||
|
ValidationSummaries: validationSummaries,
|
||||||
|
DebugDirectory: document.DebugDirectory,
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type indexDocument struct {
|
||||||
|
ManifestFile string `json:"manifest_file"`
|
||||||
|
OutputFiles *[]laneDocument `json:"output_files"`
|
||||||
|
RejectedFile string `json:"rejected_file"`
|
||||||
|
WarningsFile string `json:"warnings_file"`
|
||||||
|
DiagnosticsFile string `json:"diagnostics_file"`
|
||||||
|
ChunkMap *pipelineDocument `json:"chunk_map"`
|
||||||
|
EvidenceContext *pipelineDocument `json:"evidence_context"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type laneDocument struct {
|
||||||
|
LaneID string `json:"lane_id"`
|
||||||
|
File string `json:"file"`
|
||||||
|
MediaType string `json:"media_type"`
|
||||||
|
ModuleKey string `json:"module_key"`
|
||||||
|
SchemaID string `json:"schema_id"`
|
||||||
|
SchemaName string `json:"schema_name"`
|
||||||
|
SchemaVersion string `json:"schema_version"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type pipelineDocument struct {
|
||||||
|
ArtifactKind string `json:"artifact_kind"`
|
||||||
|
File string `json:"file"`
|
||||||
|
MediaType string `json:"media_type"`
|
||||||
|
SchemaID string `json:"schema_id"`
|
||||||
|
SchemaName string `json:"schema_name"`
|
||||||
|
SchemaVersion string `json:"schema_version"`
|
||||||
|
}
|
||||||
|
|
||||||
|
func loadIndex(bundleRoot, indexPath string) (Index, error) {
|
||||||
|
var document indexDocument
|
||||||
|
if err := decodeBoundedJSON(indexPath, maxIndexBytes, &document); err != nil {
|
||||||
|
return Index{}, fmt.Errorf("decode notarius index: %w", err)
|
||||||
|
}
|
||||||
|
for _, field := range []struct {
|
||||||
|
name string
|
||||||
|
got string
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{name: "manifest_file", got: document.ManifestFile, want: canonicalManifestFile},
|
||||||
|
{name: "rejected_file", got: document.RejectedFile, want: canonicalRejectedFile},
|
||||||
|
{name: "warnings_file", got: document.WarningsFile, want: canonicalWarningsFile},
|
||||||
|
{name: "diagnostics_file", got: document.DiagnosticsFile, want: canonicalDiagnosticsFile},
|
||||||
|
} {
|
||||||
|
if field.got != field.want {
|
||||||
|
return Index{}, fmt.Errorf("notarius index %s %q is incompatible; want %q", field.name, field.got, field.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if document.OutputFiles == nil {
|
||||||
|
return Index{}, fmt.Errorf("notarius index is missing required output_files")
|
||||||
|
}
|
||||||
|
|
||||||
|
index := Index{
|
||||||
|
Path: indexPath,
|
||||||
|
ManifestFile: document.ManifestFile,
|
||||||
|
RejectedFile: document.RejectedFile,
|
||||||
|
WarningsFile: document.WarningsFile,
|
||||||
|
DiagnosticsFile: document.DiagnosticsFile,
|
||||||
|
}
|
||||||
|
var err error
|
||||||
|
if index.ManifestPath, err = resolveRegularFile(bundleRoot, index.ManifestFile); err != nil {
|
||||||
|
return Index{}, fmt.Errorf("resolve notarius manifest file: %w", err)
|
||||||
|
}
|
||||||
|
if index.RejectedPath, err = resolveRegularFile(bundleRoot, index.RejectedFile); err != nil {
|
||||||
|
return Index{}, fmt.Errorf("resolve notarius rejection file: %w", err)
|
||||||
|
}
|
||||||
|
if index.WarningsPath, err = resolveRegularFile(bundleRoot, index.WarningsFile); err != nil {
|
||||||
|
return Index{}, fmt.Errorf("resolve notarius warning file: %w", err)
|
||||||
|
}
|
||||||
|
if index.DiagnosticsPath, err = resolveRegularFile(bundleRoot, index.DiagnosticsFile); err != nil {
|
||||||
|
return Index{}, fmt.Errorf("resolve notarius diagnostics file: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
seenLanes := make(map[string]struct{}, len(*document.OutputFiles))
|
||||||
|
for _, lane := range *document.OutputFiles {
|
||||||
|
if strings.TrimSpace(lane.LaneID) == "" || strings.TrimSpace(lane.File) == "" {
|
||||||
|
return Index{}, fmt.Errorf("notarius lane descriptors require lane_id and file")
|
||||||
|
}
|
||||||
|
if _, exists := seenLanes[lane.LaneID]; exists {
|
||||||
|
return Index{}, fmt.Errorf("notarius index contains duplicate lane id %q", lane.LaneID)
|
||||||
|
}
|
||||||
|
seenLanes[lane.LaneID] = struct{}{}
|
||||||
|
path, err := resolveRegularFile(bundleRoot, lane.File)
|
||||||
|
if err != nil {
|
||||||
|
return Index{}, fmt.Errorf("resolve notarius lane %q file: %w", lane.LaneID, err)
|
||||||
|
}
|
||||||
|
index.Lanes = append(index.Lanes, LaneDescriptor{
|
||||||
|
LaneID: lane.LaneID, File: lane.File, Path: path, MediaType: lane.MediaType,
|
||||||
|
ModuleKey: lane.ModuleKey, SchemaID: lane.SchemaID, SchemaName: lane.SchemaName,
|
||||||
|
SchemaVersion: lane.SchemaVersion,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
if document.ChunkMap != nil {
|
||||||
|
index.ChunkMap, err = resolvePipelineDescriptor(bundleRoot, "chunk_map", *document.ChunkMap)
|
||||||
|
if err != nil {
|
||||||
|
return Index{}, err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if document.EvidenceContext != nil {
|
||||||
|
index.EvidenceContext, err = resolvePipelineDescriptor(bundleRoot, "evidence_context", *document.EvidenceContext)
|
||||||
|
if err != nil {
|
||||||
|
return Index{}, err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return index, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func resolvePipelineDescriptor(bundleRoot, label string, document pipelineDocument) (*PipelineDescriptor, error) {
|
||||||
|
if strings.TrimSpace(document.ArtifactKind) == "" || strings.TrimSpace(document.File) == "" ||
|
||||||
|
strings.TrimSpace(document.MediaType) == "" || strings.TrimSpace(document.SchemaID) == "" ||
|
||||||
|
strings.TrimSpace(document.SchemaName) == "" || strings.TrimSpace(document.SchemaVersion) == "" {
|
||||||
|
return nil, fmt.Errorf("notarius %s descriptor is missing required fields", label)
|
||||||
|
}
|
||||||
|
path, err := resolveRegularFile(bundleRoot, document.File)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("resolve notarius %s file: %w", label, err)
|
||||||
|
}
|
||||||
|
return &PipelineDescriptor{
|
||||||
|
ArtifactKind: document.ArtifactKind, File: document.File, Path: path,
|
||||||
|
MediaType: document.MediaType, SchemaID: document.SchemaID,
|
||||||
|
SchemaName: document.SchemaName, SchemaVersion: document.SchemaVersion,
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type rejectionDocument struct {
|
||||||
|
Rejected *[]struct {
|
||||||
|
Stage string `json:"stage"`
|
||||||
|
StepID string `json:"step_id"`
|
||||||
|
LaneID string `json:"lane_id"`
|
||||||
|
ModuleKey string `json:"module_key"`
|
||||||
|
ChunkID string `json:"chunk_id"`
|
||||||
|
ValidatorName string `json:"validator_name"`
|
||||||
|
ReasonCode string `json:"reason_code"`
|
||||||
|
Message string `json:"message"`
|
||||||
|
} `json:"rejected"`
|
||||||
|
}
|
||||||
|
|
||||||
|
func loadRejections(path string) ([]RejectionSummary, error) {
|
||||||
|
var document rejectionDocument
|
||||||
|
if err := decodeBoundedJSON(path, maxSummaryBytes, &document); err != nil {
|
||||||
|
return nil, fmt.Errorf("decode notarius rejections: %w", err)
|
||||||
|
}
|
||||||
|
if document.Rejected == nil {
|
||||||
|
return nil, fmt.Errorf("notarius rejection document is missing rejected array")
|
||||||
|
}
|
||||||
|
summaries := make([]RejectionSummary, 0, len(*document.Rejected))
|
||||||
|
for _, item := range *document.Rejected {
|
||||||
|
if strings.TrimSpace(item.Stage) == "" || strings.TrimSpace(item.Message) == "" {
|
||||||
|
return nil, fmt.Errorf("notarius rejection entries require stage and message")
|
||||||
|
}
|
||||||
|
summaries = append(summaries, RejectionSummary{
|
||||||
|
Stage: item.Stage, StepID: item.StepID, LaneID: item.LaneID,
|
||||||
|
ModuleKey: item.ModuleKey, ChunkID: item.ChunkID,
|
||||||
|
ValidatorName: item.ValidatorName, ReasonCode: item.ReasonCode,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
return summaries, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type findingGroupDocument struct {
|
||||||
|
Disposition string `json:"disposition"`
|
||||||
|
Category string `json:"category"`
|
||||||
|
ReasonCode string `json:"reason_code"`
|
||||||
|
Origin diagnosticOriginDocument `json:"origin"`
|
||||||
|
OccurrenceCount *int `json:"occurrence_count"`
|
||||||
|
Samples *[]struct {
|
||||||
|
Scope string `json:"scope"`
|
||||||
|
Message string `json:"message"`
|
||||||
|
ChunkID string `json:"chunk_id"`
|
||||||
|
ChunkIndex *int `json:"chunk_index"`
|
||||||
|
} `json:"samples"`
|
||||||
|
OmittedSampleCount *int `json:"omitted_sample_count"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type diagnosticOriginDocument struct {
|
||||||
|
Stage string `json:"stage"`
|
||||||
|
StepID string `json:"step_id"`
|
||||||
|
LaneID string `json:"lane_id"`
|
||||||
|
ModuleKey string `json:"module_key"`
|
||||||
|
ValidatorKey string `json:"validator_key"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type warningDocument struct {
|
||||||
|
SchemaVersion string `json:"schema_version"`
|
||||||
|
GroupCount *int `json:"group_count"`
|
||||||
|
OccurrenceCount *int `json:"occurrence_count"`
|
||||||
|
Groups *[]findingGroupDocument `json:"groups"`
|
||||||
|
}
|
||||||
|
|
||||||
|
func loadWarnings(path string) ([]WarningSummary, error) {
|
||||||
|
var document warningDocument
|
||||||
|
if err := decodeBoundedJSON(path, maxSummaryBytes, &document); err != nil {
|
||||||
|
return nil, fmt.Errorf("decode notarius warnings: %w", err)
|
||||||
|
}
|
||||||
|
if document.SchemaVersion != warningsSchemaVersion || document.GroupCount == nil ||
|
||||||
|
document.OccurrenceCount == nil || document.Groups == nil {
|
||||||
|
return nil, fmt.Errorf("notarius warning document is missing or incompatible required fields")
|
||||||
|
}
|
||||||
|
if *document.GroupCount < 0 || *document.GroupCount > maxWarningGroups || *document.OccurrenceCount < 0 ||
|
||||||
|
*document.GroupCount != len(*document.Groups) {
|
||||||
|
return nil, fmt.Errorf("notarius warning document counts are inconsistent")
|
||||||
|
}
|
||||||
|
summaries := make([]WarningSummary, 0, len(*document.Groups))
|
||||||
|
occurrences := 0
|
||||||
|
for _, group := range *document.Groups {
|
||||||
|
if err := validateFindingGroup(group); err != nil {
|
||||||
|
return nil, fmt.Errorf("notarius warning group: %w", err)
|
||||||
|
}
|
||||||
|
if group.Disposition != "warning" {
|
||||||
|
return nil, fmt.Errorf("notarius warning group disposition %q is invalid", group.Disposition)
|
||||||
|
}
|
||||||
|
if *group.OccurrenceCount > int(^uint(0)>>1)-occurrences {
|
||||||
|
return nil, fmt.Errorf("notarius warning occurrence count overflows")
|
||||||
|
}
|
||||||
|
occurrences += *group.OccurrenceCount
|
||||||
|
summaries = append(summaries, WarningSummary{
|
||||||
|
Disposition: group.Disposition, Category: group.Category, ReasonCode: group.ReasonCode,
|
||||||
|
Origin: diagnosticOrigin(group.Origin), OccurrenceCount: *group.OccurrenceCount,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
if occurrences != *document.OccurrenceCount {
|
||||||
|
return nil, fmt.Errorf("notarius warning document occurrence count is inconsistent")
|
||||||
|
}
|
||||||
|
return summaries, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type diagnosticDocument struct {
|
||||||
|
SchemaVersion string `json:"schema_version"`
|
||||||
|
GroupCount *int `json:"group_count"`
|
||||||
|
OccurrenceCount *int `json:"occurrence_count"`
|
||||||
|
Truncated *bool `json:"truncated"`
|
||||||
|
UnrepresentedOccurrenceCount *int `json:"unrepresented_occurrence_count"`
|
||||||
|
Groups *[]findingGroupDocument `json:"groups"`
|
||||||
|
}
|
||||||
|
|
||||||
|
func loadDiagnostics(path string) ([]DiagnosticSummary, int, bool, error) {
|
||||||
|
var document diagnosticDocument
|
||||||
|
if err := decodeBoundedJSON(path, maxSummaryBytes, &document); err != nil {
|
||||||
|
return nil, 0, false, fmt.Errorf("decode notarius diagnostics: %w", err)
|
||||||
|
}
|
||||||
|
if document.SchemaVersion != diagnosticsSchemaVersion || document.GroupCount == nil ||
|
||||||
|
document.OccurrenceCount == nil || document.Truncated == nil ||
|
||||||
|
document.UnrepresentedOccurrenceCount == nil || document.Groups == nil {
|
||||||
|
return nil, 0, false, fmt.Errorf("notarius diagnostics document is missing or incompatible required fields")
|
||||||
|
}
|
||||||
|
if *document.GroupCount < 0 || *document.GroupCount > maxDiagnosticGroups || *document.OccurrenceCount < 0 ||
|
||||||
|
*document.UnrepresentedOccurrenceCount < 0 || *document.GroupCount != len(*document.Groups) {
|
||||||
|
return nil, 0, false, fmt.Errorf("notarius diagnostics document counts are inconsistent")
|
||||||
|
}
|
||||||
|
if !*document.Truncated && *document.UnrepresentedOccurrenceCount != 0 {
|
||||||
|
return nil, 0, false, fmt.Errorf("notarius diagnostics document has unrepresented occurrences without truncation")
|
||||||
|
}
|
||||||
|
summaries := make([]DiagnosticSummary, 0, len(*document.Groups))
|
||||||
|
representedOccurrences := 0
|
||||||
|
for _, group := range *document.Groups {
|
||||||
|
if err := validateFindingGroup(group); err != nil {
|
||||||
|
return nil, 0, false, fmt.Errorf("notarius diagnostic group: %w", err)
|
||||||
|
}
|
||||||
|
if group.Disposition != "advisory" && group.Disposition != "observation" {
|
||||||
|
return nil, 0, false, fmt.Errorf("notarius diagnostic group disposition %q is invalid", group.Disposition)
|
||||||
|
}
|
||||||
|
if *group.OccurrenceCount > int(^uint(0)>>1)-representedOccurrences {
|
||||||
|
return nil, 0, false, fmt.Errorf("notarius diagnostic occurrence count overflows")
|
||||||
|
}
|
||||||
|
representedOccurrences += *group.OccurrenceCount
|
||||||
|
summaries = append(summaries, DiagnosticSummary{
|
||||||
|
Disposition: group.Disposition, Category: group.Category, ReasonCode: group.ReasonCode,
|
||||||
|
Origin: diagnosticOrigin(group.Origin), OccurrenceCount: *group.OccurrenceCount,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
if *document.UnrepresentedOccurrenceCount > int(^uint(0)>>1)-representedOccurrences ||
|
||||||
|
representedOccurrences+*document.UnrepresentedOccurrenceCount != *document.OccurrenceCount {
|
||||||
|
return nil, 0, false, fmt.Errorf("notarius diagnostics document occurrence count is inconsistent")
|
||||||
|
}
|
||||||
|
return summaries, *document.OccurrenceCount, *document.Truncated, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func validateFindingGroup(group findingGroupDocument) error {
|
||||||
|
if strings.TrimSpace(group.Disposition) == "" || strings.TrimSpace(group.Category) == "" ||
|
||||||
|
strings.TrimSpace(group.ReasonCode) == "" || !validDiagnosticOriginStage(group.Origin.Stage) ||
|
||||||
|
!validDiagnosticCategory(group.Disposition, group.Category) ||
|
||||||
|
group.OccurrenceCount == nil || *group.OccurrenceCount <= 0 || group.Samples == nil ||
|
||||||
|
group.OmittedSampleCount == nil || *group.OmittedSampleCount < 0 {
|
||||||
|
return fmt.Errorf("missing required fields")
|
||||||
|
}
|
||||||
|
if len(*group.Samples) == 0 || len(*group.Samples) > maxFindingSamples ||
|
||||||
|
*group.OmittedSampleCount != *group.OccurrenceCount-len(*group.Samples) {
|
||||||
|
return fmt.Errorf("sample counts are inconsistent")
|
||||||
|
}
|
||||||
|
for _, sample := range *group.Samples {
|
||||||
|
if strings.TrimSpace(sample.Scope) == "" || strings.TrimSpace(sample.Message) == "" ||
|
||||||
|
(sample.ChunkIndex != nil && *sample.ChunkIndex < 0) {
|
||||||
|
return fmt.Errorf("samples require scope and message")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func validDiagnosticCategory(disposition, category string) bool {
|
||||||
|
switch disposition {
|
||||||
|
case "warning":
|
||||||
|
return category == "configuration" || category == "degradation" ||
|
||||||
|
category == "validation_incomplete" || category == "fallback"
|
||||||
|
case "advisory":
|
||||||
|
return category == "data_quality"
|
||||||
|
case "observation":
|
||||||
|
return category == "normalization"
|
||||||
|
default:
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func validDiagnosticOriginStage(stage string) bool {
|
||||||
|
switch stage {
|
||||||
|
case "references", "chunk", "extract", "merge", "normalize":
|
||||||
|
return true
|
||||||
|
default:
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func diagnosticOrigin(document diagnosticOriginDocument) DiagnosticOrigin {
|
||||||
|
return DiagnosticOrigin{
|
||||||
|
Stage: document.Stage, StepID: document.StepID, LaneID: document.LaneID,
|
||||||
|
ModuleKey: document.ModuleKey, ValidatorKey: document.ValidatorKey,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func sumWarningOccurrences(values []WarningSummary) (int, error) {
|
||||||
|
total := 0
|
||||||
|
for _, value := range values {
|
||||||
|
if value.OccurrenceCount > int(^uint(0)>>1)-total {
|
||||||
|
return 0, fmt.Errorf("notarius warning occurrence count overflows")
|
||||||
|
}
|
||||||
|
total += value.OccurrenceCount
|
||||||
|
}
|
||||||
|
return total, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func decodeBoundedJSON(path string, limit int64, destination any) error {
|
||||||
|
data, err := fileops.ReadRegularFile(path, limit)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("notarius JSON result exceeds or cannot be read within %d-byte limit: %w", limit, err)
|
||||||
|
}
|
||||||
|
if err := json.Unmarshal(data, destination); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func validateBundleRoot(outputRoot, bundleRoot string) (string, error) {
|
||||||
|
root := filepath.Clean(outputRoot)
|
||||||
|
bundle := filepath.Clean(bundleRoot)
|
||||||
|
relative, err := filepath.Rel(root, bundle)
|
||||||
|
if err != nil {
|
||||||
|
return "", fmt.Errorf("compare notarius output paths: %w", err)
|
||||||
|
}
|
||||||
|
if relative == "." || relative == ".." || strings.HasPrefix(relative, ".."+string(filepath.Separator)) {
|
||||||
|
return "", fmt.Errorf("notarius output directory %q is not beneath output root %q", bundleRoot, outputRoot)
|
||||||
|
}
|
||||||
|
if err := requireDirectoryTree(root, relative); err != nil {
|
||||||
|
return "", fmt.Errorf("validate notarius output directory: %w", err)
|
||||||
|
}
|
||||||
|
return bundle, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func resolveRegularFile(root, logicalPath string) (string, error) {
|
||||||
|
resolved, err := pathsafe.JoinSlashRelativeUnderRoot(root, logicalPath)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
relative, err := filepath.Rel(root, resolved)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
if err := requireRegularFileTree(root, relative); err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
return resolved, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func requireDirectoryTree(root, relative string) error {
|
||||||
|
if err := requireDirectory(root); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
current := root
|
||||||
|
for _, component := range strings.Split(relative, string(filepath.Separator)) {
|
||||||
|
current = filepath.Join(current, component)
|
||||||
|
if err := requireDirectory(current); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func requireRegularFileTree(root, relative string) error {
|
||||||
|
components := strings.Split(relative, string(filepath.Separator))
|
||||||
|
if len(components) == 0 {
|
||||||
|
return fmt.Errorf("regular file path is required")
|
||||||
|
}
|
||||||
|
if err := requireDirectory(root); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
current := root
|
||||||
|
for _, component := range components[:len(components)-1] {
|
||||||
|
current = filepath.Join(current, component)
|
||||||
|
if err := requireDirectory(current); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return requireRegularFile(filepath.Join(current, components[len(components)-1]))
|
||||||
|
}
|
||||||
|
|
||||||
|
func requireDirectory(path string) error {
|
||||||
|
info, err := os.Lstat(path)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if info.Mode()&os.ModeSymlink != 0 || !info.IsDir() {
|
||||||
|
return fmt.Errorf("path %q must be a directory without symlinks", path)
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func requireRegularFile(path string) error {
|
||||||
|
info, err := os.Lstat(path)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if info.Mode()&os.ModeSymlink != 0 || !info.Mode().IsRegular() {
|
||||||
|
return fmt.Errorf("path %q must be a regular file without symlinks", path)
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func validateLogDestination(path string) error {
|
||||||
|
if err := requireDirectory(filepath.Dir(path)); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
info, err := os.Lstat(path)
|
||||||
|
if errors.Is(err, os.ErrNotExist) {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if info.Mode()&os.ModeSymlink != 0 || !info.Mode().IsRegular() {
|
||||||
|
return fmt.Errorf("path %q must be absent or a regular file without symlinks", path)
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
716
internal/adapters/notarius/subprocess_test.go
Normal file
716
internal/adapters/notarius/subprocess_test.go
Normal file
@@ -0,0 +1,716 @@
|
|||||||
|
package notarius
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"errors"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"reflect"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
sharedsubprocess "gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestSubprocessRunnerBuildsExactInvocationAndDiscoversBundle(t *testing.T) {
|
||||||
|
req := validRunRequest(t)
|
||||||
|
var captured sharedsubprocess.RunRequest
|
||||||
|
runner := &SubprocessRunner{run: func(_ context.Context, processReq sharedsubprocess.RunRequest) (sharedsubprocess.RunResult, error) {
|
||||||
|
captured = processReq
|
||||||
|
writeValidBundleAndReceipt(t, req, true)
|
||||||
|
return sharedsubprocess.RunResult{ExitCode: 0, Duration: 2 * time.Second}, nil
|
||||||
|
}}
|
||||||
|
|
||||||
|
result, err := runner.Run(context.Background(), req)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Run() error = %v", err)
|
||||||
|
}
|
||||||
|
wantArgs := []string{
|
||||||
|
"run", "dnd-session", "--config", req.ConfigPath, "--input", req.InputPath,
|
||||||
|
"--output-dir", req.OutputRoot, "--json",
|
||||||
|
}
|
||||||
|
if !reflect.DeepEqual(captured.Args, wantArgs) {
|
||||||
|
t.Fatalf("subprocess args = %#v, want %#v", captured.Args, wantArgs)
|
||||||
|
}
|
||||||
|
if captured.Executable != req.Binary || captured.WorkingDir != req.WorkingDirectory || captured.Timeout != req.Timeout {
|
||||||
|
t.Fatalf("subprocess request = %#v", captured)
|
||||||
|
}
|
||||||
|
if captured.StdoutLogPath != req.ReceiptPath || captured.StderrLogPath != req.LogPath {
|
||||||
|
t.Fatalf("stream paths = stdout %q stderr %q", captured.StdoutLogPath, captured.StderrLogPath)
|
||||||
|
}
|
||||||
|
if captured.EnvOverrides != nil {
|
||||||
|
t.Fatalf("environment overrides = %#v, want inherited environment only", captured.EnvOverrides)
|
||||||
|
}
|
||||||
|
for _, arg := range captured.Args {
|
||||||
|
if arg == "--session-id" {
|
||||||
|
t.Fatal("subprocess args unexpectedly contain --session-id")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if result.Receipt.SchemaVersion != ReceiptSchemaVersion || result.Receipt.RunID != "notarius-run-1" {
|
||||||
|
t.Fatalf("receipt = %#v", result.Receipt)
|
||||||
|
}
|
||||||
|
if len(result.Receipt.ValidationSummaries) != 1 || result.Receipt.ValidationSummaries[0].LaneID != "npc-registry" ||
|
||||||
|
result.Receipt.ValidationSummaries[0].Status != "complete" {
|
||||||
|
t.Fatalf("validation summaries = %#v", result.Receipt.ValidationSummaries)
|
||||||
|
}
|
||||||
|
if len(result.Index.Lanes) != 1 || result.Index.Lanes[0].LaneID != "npc-registry" {
|
||||||
|
t.Fatalf("lanes = %#v", result.Index.Lanes)
|
||||||
|
}
|
||||||
|
if result.Index.ChunkMap == nil || result.Index.ChunkMap.ArtifactKind != "chunk_map" {
|
||||||
|
t.Fatalf("chunk map = %#v", result.Index.ChunkMap)
|
||||||
|
}
|
||||||
|
if result.Index.EvidenceContext == nil || result.Index.EvidenceContext.ArtifactKind != "evidence_context" {
|
||||||
|
t.Fatalf("evidence context = %#v", result.Index.EvidenceContext)
|
||||||
|
}
|
||||||
|
if len(result.Rejections) != 1 || result.Rejections[0].LaneID != "spells" || result.Rejections[0].ReasonCode != "invalid_spell" {
|
||||||
|
t.Fatalf("rejections = %#v", result.Rejections)
|
||||||
|
}
|
||||||
|
if len(result.Warnings) != 1 || result.Warnings[0].Category != "degradation" || result.Warnings[0].ReasonCode != "normalized_name" {
|
||||||
|
t.Fatalf("warnings = %#v", result.Warnings)
|
||||||
|
}
|
||||||
|
if len(result.Diagnostics) != 1 || result.Diagnostics[0].Category != "data_quality" || result.Diagnostics[0].ReasonCode != "low_confidence" {
|
||||||
|
t.Fatalf("diagnostics = %#v", result.Diagnostics)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSubprocessRunnerBuildsOrderedReferenceArguments(t *testing.T) {
|
||||||
|
req := validRunRequest(t)
|
||||||
|
referenceRoot := t.TempDir()
|
||||||
|
req.References = []ReferenceBinding{
|
||||||
|
{Selector: " party ", Path: filepath.Join(referenceRoot, "party context=primary.json")},
|
||||||
|
{Selector: " npc-registry . extract . glossary ", Path: filepath.Join(referenceRoot, "glossary.json")},
|
||||||
|
}
|
||||||
|
originalReferences := append([]ReferenceBinding(nil), req.References...)
|
||||||
|
var captured sharedsubprocess.RunRequest
|
||||||
|
runner := &SubprocessRunner{run: func(_ context.Context, processReq sharedsubprocess.RunRequest) (sharedsubprocess.RunResult, error) {
|
||||||
|
captured = processReq
|
||||||
|
writeValidBundleAndReceipt(t, req, false)
|
||||||
|
return sharedsubprocess.RunResult{ExitCode: 0}, nil
|
||||||
|
}}
|
||||||
|
|
||||||
|
if _, err := runner.Run(context.Background(), req); err != nil {
|
||||||
|
t.Fatalf("Run() error = %v", err)
|
||||||
|
}
|
||||||
|
wantArgs := []string{
|
||||||
|
"run", req.PipelineID,
|
||||||
|
"--config", req.ConfigPath,
|
||||||
|
"--input", req.InputPath,
|
||||||
|
"--output-dir", req.OutputRoot,
|
||||||
|
"--reference", "party=" + req.References[0].Path,
|
||||||
|
"--reference", "npc-registry.extract.glossary=" + req.References[1].Path,
|
||||||
|
"--json",
|
||||||
|
}
|
||||||
|
if !reflect.DeepEqual(captured.Args, wantArgs) {
|
||||||
|
t.Fatalf("subprocess args = %#v, want %#v", captured.Args, wantArgs)
|
||||||
|
}
|
||||||
|
if !reflect.DeepEqual(req.References, originalReferences) {
|
||||||
|
t.Fatalf("Run() mutated caller references = %#v, want %#v", req.References, originalReferences)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSubprocessRunnerRejectsInvalidReferencesBeforeLaunch(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
references func(string) []ReferenceBinding
|
||||||
|
wantErr string
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
name: "invalid selector",
|
||||||
|
references: func(root string) []ReferenceBinding {
|
||||||
|
return []ReferenceBinding{{Selector: "lane.prepare.party", Path: filepath.Join(root, "party.json")}}
|
||||||
|
},
|
||||||
|
wantErr: "selector",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "duplicate normalized selector",
|
||||||
|
references: func(root string) []ReferenceBinding {
|
||||||
|
return []ReferenceBinding{
|
||||||
|
{Selector: "lane.party", Path: filepath.Join(root, "party.json")},
|
||||||
|
{Selector: " lane . party ", Path: filepath.Join(root, "party-2.json")},
|
||||||
|
}
|
||||||
|
},
|
||||||
|
wantErr: "duplicated",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "empty path",
|
||||||
|
references: func(string) []ReferenceBinding {
|
||||||
|
return []ReferenceBinding{{Selector: "party", Path: " "}}
|
||||||
|
},
|
||||||
|
wantErr: "path is required",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "relative path",
|
||||||
|
references: func(string) []ReferenceBinding {
|
||||||
|
return []ReferenceBinding{{Selector: "party", Path: "references/party.json"}}
|
||||||
|
},
|
||||||
|
wantErr: "path must be absolute",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
req := validRunRequest(t)
|
||||||
|
req.References = tt.references(t.TempDir())
|
||||||
|
started := false
|
||||||
|
runner := &SubprocessRunner{run: func(context.Context, sharedsubprocess.RunRequest) (sharedsubprocess.RunResult, error) {
|
||||||
|
started = true
|
||||||
|
return sharedsubprocess.RunResult{}, nil
|
||||||
|
}}
|
||||||
|
|
||||||
|
_, err := runner.Run(context.Background(), req)
|
||||||
|
if err == nil || !strings.Contains(err.Error(), tt.wantErr) {
|
||||||
|
t.Fatalf("Run() error = %v, want containing %q", err, tt.wantErr)
|
||||||
|
}
|
||||||
|
if started {
|
||||||
|
t.Fatal("subprocess started after request validation failure")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSubprocessRunnerUsesMinimalEnvironmentAndSeparatesStreams(t *testing.T) {
|
||||||
|
req := validRunRequest(t)
|
||||||
|
writeValidBundleAndReceipt(t, req, false)
|
||||||
|
receiptFixture := req.ReceiptPath + ".fixture"
|
||||||
|
data, err := os.ReadFile(req.ReceiptPath)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ReadFile(receipt) error = %v", err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(receiptFixture, data, 0o644); err != nil {
|
||||||
|
t.Fatalf("WriteFile(receipt fixture) error = %v", err)
|
||||||
|
}
|
||||||
|
if err := os.Remove(req.ReceiptPath); err != nil {
|
||||||
|
t.Fatalf("Remove(receipt) error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
captureDir := filepath.Join(filepath.Dir(req.ReceiptPath), "capture")
|
||||||
|
if err := os.Mkdir(captureDir, 0o755); err != nil {
|
||||||
|
t.Fatalf("Mkdir(capture) error = %v", err)
|
||||||
|
}
|
||||||
|
script := writeShellScript(t, `#!/bin/sh
|
||||||
|
pwd > "$NOTARIUS_CAPTURE_DIR/working-directory"
|
||||||
|
printf '%s' "$NOTARIUS_INHERITED_VALUE" > "$NOTARIUS_CAPTURE_DIR/environment"
|
||||||
|
printf 'diagnostic stream\n' >&2
|
||||||
|
cat "$NOTARIUS_RECEIPT_FIXTURE"
|
||||||
|
`)
|
||||||
|
req.Binary = script
|
||||||
|
t.Setenv("NOTARIUS_CAPTURE_DIR", captureDir)
|
||||||
|
t.Setenv("NOTARIUS_INHERITED_VALUE", "inherited-value")
|
||||||
|
t.Setenv("NOTARIUS_RECEIPT_FIXTURE", receiptFixture)
|
||||||
|
|
||||||
|
if _, err := NewSubprocessRunner().Run(context.Background(), req); err != nil {
|
||||||
|
t.Fatalf("Run() error = %v", err)
|
||||||
|
}
|
||||||
|
assertTextFile(t, filepath.Join(captureDir, "working-directory"), req.WorkingDirectory+"\n")
|
||||||
|
assertTextFile(t, filepath.Join(captureDir, "environment"), "")
|
||||||
|
assertTextFile(t, req.LogPath, "diagnostic stream\n")
|
||||||
|
receiptBytes, err := os.ReadFile(req.ReceiptPath)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ReadFile(receipt) error = %v", err)
|
||||||
|
}
|
||||||
|
if strings.Contains(string(receiptBytes), "diagnostic stream") {
|
||||||
|
t.Fatal("receipt contains stderr output")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSubprocessRunnerReturnsProcessFailuresWithoutParsingStdout(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
scriptBody string
|
||||||
|
timeout time.Duration
|
||||||
|
cancel bool
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{name: "nonzero", scriptBody: "printf '{malformed receipt'; printf 'failed\\n' >&2; exit 7\n", timeout: time.Second, want: "exit code 7"},
|
||||||
|
{name: "timeout", scriptBody: "sleep 5\n", timeout: 20 * time.Millisecond, want: "timed out"},
|
||||||
|
{name: "cancellation", scriptBody: "sleep 5\n", timeout: time.Second, cancel: true, want: "canceled"},
|
||||||
|
}
|
||||||
|
for _, test := range tests {
|
||||||
|
t.Run(test.name, func(t *testing.T) {
|
||||||
|
req := validRunRequest(t)
|
||||||
|
req.Binary = writeShellScript(t, "#!/bin/sh\n"+test.scriptBody)
|
||||||
|
req.Timeout = test.timeout
|
||||||
|
ctx := context.Background()
|
||||||
|
if test.cancel {
|
||||||
|
cancelCtx, cancel := context.WithCancel(ctx)
|
||||||
|
ctx = cancelCtx
|
||||||
|
time.AfterFunc(20*time.Millisecond, cancel)
|
||||||
|
}
|
||||||
|
_, err := NewSubprocessRunner().Run(ctx, req)
|
||||||
|
if err == nil || !strings.Contains(err.Error(), test.want) {
|
||||||
|
t.Fatalf("Run() error = %v, want fragment %q", err, test.want)
|
||||||
|
}
|
||||||
|
if strings.Contains(err.Error(), "decode notarius receipt") {
|
||||||
|
t.Fatalf("Run() parsed stdout after process failure: %v", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSubprocessRunnerReturnsSharedSubprocessErrorWithoutReadingReceipt(t *testing.T) {
|
||||||
|
req := validRunRequest(t)
|
||||||
|
if err := os.WriteFile(req.ReceiptPath, []byte("not json"), 0o644); err != nil {
|
||||||
|
t.Fatalf("WriteFile(receipt) error = %v", err)
|
||||||
|
}
|
||||||
|
wantErr := errors.New("process failed")
|
||||||
|
runner := &SubprocessRunner{run: func(context.Context, sharedsubprocess.RunRequest) (sharedsubprocess.RunResult, error) {
|
||||||
|
return sharedsubprocess.RunResult{ExitCode: 9}, wantErr
|
||||||
|
}}
|
||||||
|
_, err := runner.Run(context.Background(), req)
|
||||||
|
if !errors.Is(err, wantErr) {
|
||||||
|
t.Fatalf("Run() error = %v, want wrapped process error", err)
|
||||||
|
}
|
||||||
|
if strings.Contains(err.Error(), "decode") {
|
||||||
|
t.Fatalf("Run() parsed receipt after failure: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLoadReceiptValidation(t *testing.T) {
|
||||||
|
root := t.TempDir()
|
||||||
|
valid := map[string]any{
|
||||||
|
"schema_version": ReceiptSchemaVersion, "run_id": "run-1", "pipeline_id": "pipeline-1",
|
||||||
|
"output_directory": filepath.Join(root, "outputs", "run-1"), "index_file": "index.json",
|
||||||
|
"normalized_output_count": 1, "rejected_output_count": 0,
|
||||||
|
"warning_group_count": 0, "warning_occurrence_count": 0,
|
||||||
|
"diagnostic_group_count": 0, "diagnostic_occurrence_count": 0,
|
||||||
|
"diagnostics_truncated": false, "validation_status": "approved", "future_field": true,
|
||||||
|
}
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
mutate func(map[string]any)
|
||||||
|
raw []byte
|
||||||
|
wantOK bool
|
||||||
|
wantError string
|
||||||
|
}{
|
||||||
|
{name: "unknown fields tolerated", wantOK: true},
|
||||||
|
{name: "malformed", raw: []byte("{")},
|
||||||
|
{name: "unsupported version", mutate: func(v map[string]any) { v["schema_version"] = "notarius.run-result.v1" }},
|
||||||
|
{name: "missing field", mutate: func(v map[string]any) { delete(v, "run_id") }},
|
||||||
|
{name: "pipeline mismatch", mutate: func(v map[string]any) { v["pipeline_id"] = "other" }},
|
||||||
|
{name: "relative output", mutate: func(v map[string]any) { v["output_directory"] = "run-1" }},
|
||||||
|
{name: "negative count", mutate: func(v map[string]any) { v["warning_group_count"] = -1 }},
|
||||||
|
{name: "invalid validation status", mutate: func(v map[string]any) { v["validation_status"] = "valid" }},
|
||||||
|
{
|
||||||
|
name: "nested index", mutate: func(v map[string]any) { v["index_file"] = "nested/index.json" },
|
||||||
|
wantError: `index_file "nested/index.json"`,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "cleanable index", mutate: func(v map[string]any) { v["index_file"] = "./index.json" },
|
||||||
|
wantError: `index_file "./index.json"`,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
for _, test := range tests {
|
||||||
|
t.Run(test.name, func(t *testing.T) {
|
||||||
|
path := filepath.Join(root, strings.ReplaceAll(test.name, " ", "-")+".json")
|
||||||
|
values := cloneMap(valid)
|
||||||
|
if test.mutate != nil {
|
||||||
|
test.mutate(values)
|
||||||
|
}
|
||||||
|
if test.raw != nil {
|
||||||
|
if err := os.WriteFile(path, test.raw, 0o644); err != nil {
|
||||||
|
t.Fatalf("WriteFile() error = %v", err)
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
writeJSONFile(t, path, values)
|
||||||
|
}
|
||||||
|
_, err := loadReceipt(path, "pipeline-1")
|
||||||
|
if test.wantOK && err != nil {
|
||||||
|
t.Fatalf("loadReceipt() error = %v", err)
|
||||||
|
}
|
||||||
|
if !test.wantOK && err == nil {
|
||||||
|
t.Fatal("loadReceipt() error = nil, want validation failure")
|
||||||
|
}
|
||||||
|
if test.wantError != "" && !strings.Contains(err.Error(), test.wantError) {
|
||||||
|
t.Fatalf("loadReceipt() error = %v, want fragment %q", err, test.wantError)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
oversized := filepath.Join(root, "oversized.json")
|
||||||
|
if err := os.WriteFile(oversized, []byte(strings.Repeat("x", maxReceiptBytes+1)), 0o644); err != nil {
|
||||||
|
t.Fatalf("WriteFile(oversized) error = %v", err)
|
||||||
|
}
|
||||||
|
if _, err := loadReceipt(oversized, "pipeline-1"); err == nil || !strings.Contains(err.Error(), "exceeds") {
|
||||||
|
t.Fatalf("loadReceipt(oversized) error = %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestValidateBundleRootRejectsEscapesAndSymlinks(t *testing.T) {
|
||||||
|
root := t.TempDir()
|
||||||
|
outputRoot := filepath.Join(root, "output")
|
||||||
|
if err := os.Mkdir(outputRoot, 0o755); err != nil {
|
||||||
|
t.Fatalf("Mkdir(output root) error = %v", err)
|
||||||
|
}
|
||||||
|
validBundle := filepath.Join(outputRoot, "run-1")
|
||||||
|
if err := os.Mkdir(validBundle, 0o755); err != nil {
|
||||||
|
t.Fatalf("Mkdir(bundle) error = %v", err)
|
||||||
|
}
|
||||||
|
if _, err := validateBundleRoot(outputRoot, validBundle); err != nil {
|
||||||
|
t.Fatalf("validateBundleRoot(valid) error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
outside := filepath.Join(root, "output-other")
|
||||||
|
if err := os.Mkdir(outside, 0o755); err != nil {
|
||||||
|
t.Fatalf("Mkdir(outside) error = %v", err)
|
||||||
|
}
|
||||||
|
for name, candidate := range map[string]string{"equal root": outputRoot, "escape": root, "prefix confusion": outside} {
|
||||||
|
t.Run(name, func(t *testing.T) {
|
||||||
|
if _, err := validateBundleRoot(outputRoot, candidate); err == nil {
|
||||||
|
t.Fatalf("validateBundleRoot(%q) error = nil", candidate)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
symlink := filepath.Join(outputRoot, "linked")
|
||||||
|
if err := os.Symlink(outside, symlink); err != nil {
|
||||||
|
t.Skipf("Symlink() unavailable: %v", err)
|
||||||
|
}
|
||||||
|
if _, err := validateBundleRoot(outputRoot, symlink); err == nil {
|
||||||
|
t.Fatal("validateBundleRoot(symlink) error = nil")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLoadIndexRejectsMalformedUnsafeAndUnsupportedDocuments(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
indexValue any
|
||||||
|
prepare func(*testing.T, string)
|
||||||
|
wantError string
|
||||||
|
}{
|
||||||
|
{name: "malformed", indexValue: json.RawMessage(`{"manifest_file":`)},
|
||||||
|
{name: "unsupported output shape", indexValue: map[string]any{"manifest_file": "manifest.json", "output_files": map[string]any{}, "rejected_file": "rejected.json", "warnings_file": "warnings.json"}},
|
||||||
|
{name: "missing management path", indexValue: map[string]any{"output_files": []any{}, "rejected_file": "rejected.json", "warnings_file": "warnings.json"}},
|
||||||
|
{name: "renamed manifest", indexValue: func() any {
|
||||||
|
value := validIndexValue([]any{})
|
||||||
|
value["manifest_file"] = "metadata.json"
|
||||||
|
return value
|
||||||
|
}(), wantError: `manifest_file "metadata.json"`},
|
||||||
|
{name: "cleanable manifest", indexValue: func() any {
|
||||||
|
value := validIndexValue([]any{})
|
||||||
|
value["manifest_file"] = "./manifest.json"
|
||||||
|
return value
|
||||||
|
}(), wantError: `manifest_file "./manifest.json"`},
|
||||||
|
{name: "renamed rejections", indexValue: func() any {
|
||||||
|
value := validIndexValue([]any{})
|
||||||
|
value["rejected_file"] = "rejections.json"
|
||||||
|
return value
|
||||||
|
}(), wantError: `rejected_file "rejections.json"`},
|
||||||
|
{name: "renamed warnings", indexValue: func() any {
|
||||||
|
value := validIndexValue([]any{})
|
||||||
|
value["warnings_file"] = "diagnostics/warnings.json"
|
||||||
|
return value
|
||||||
|
}(), wantError: `warnings_file "diagnostics/warnings.json"`},
|
||||||
|
{name: "duplicate lane", indexValue: validIndexValue([]any{
|
||||||
|
map[string]any{"lane_id": "npc", "file": "lanes/npc.json"},
|
||||||
|
map[string]any{"lane_id": "npc", "file": "lanes/npc.json"},
|
||||||
|
})},
|
||||||
|
{name: "absolute logical path", indexValue: validIndexValue([]any{map[string]any{"lane_id": "npc", "file": "/tmp/npc.json"}})},
|
||||||
|
{name: "lexical traversal", indexValue: validIndexValue([]any{map[string]any{"lane_id": "npc", "file": "../outside.json"}})},
|
||||||
|
{name: "root prefix confusion", indexValue: validIndexValue([]any{map[string]any{"lane_id": "npc", "file": "../bundle-other/npc.json"}})},
|
||||||
|
{name: "file symlink", indexValue: validIndexValue([]any{map[string]any{"lane_id": "npc", "file": "lanes/npc.json"}}), prepare: func(t *testing.T, bundle string) {
|
||||||
|
if err := os.Symlink(filepath.Join(bundle, "manifest.json"), filepath.Join(bundle, "lanes", "npc.json")); err != nil {
|
||||||
|
t.Skipf("Symlink() unavailable: %v", err)
|
||||||
|
}
|
||||||
|
}},
|
||||||
|
{name: "directory symlink", indexValue: validIndexValue([]any{map[string]any{"lane_id": "npc", "file": "linked/npc.json"}}), prepare: func(t *testing.T, bundle string) {
|
||||||
|
if err := os.Symlink(filepath.Join(bundle, "lanes"), filepath.Join(bundle, "linked")); err != nil {
|
||||||
|
t.Skipf("Symlink() unavailable: %v", err)
|
||||||
|
}
|
||||||
|
}},
|
||||||
|
{name: "missing management file", indexValue: validIndexValue([]any{}), prepare: func(t *testing.T, bundle string) {
|
||||||
|
if err := os.Remove(filepath.Join(bundle, "manifest.json")); err != nil {
|
||||||
|
t.Fatalf("Remove(manifest) error = %v", err)
|
||||||
|
}
|
||||||
|
}},
|
||||||
|
{name: "incomplete pipeline descriptor", indexValue: func() any {
|
||||||
|
value := validIndexValue([]any{})
|
||||||
|
value["chunk_map"] = map[string]any{"artifact_kind": "chunk_map", "file": "chunk-map.json"}
|
||||||
|
return value
|
||||||
|
}()},
|
||||||
|
{name: "pipeline descriptor escape", indexValue: func() any {
|
||||||
|
value := validIndexValue([]any{})
|
||||||
|
value["evidence_context"] = map[string]any{
|
||||||
|
"artifact_kind": "evidence_context", "file": "../evidence.json", "media_type": "application/json",
|
||||||
|
"schema_id": "evidence", "schema_name": "Evidence", "schema_version": "v1",
|
||||||
|
}
|
||||||
|
return value
|
||||||
|
}()},
|
||||||
|
}
|
||||||
|
for _, test := range tests {
|
||||||
|
t.Run(test.name, func(t *testing.T) {
|
||||||
|
bundle := createBundleSkeleton(t)
|
||||||
|
indexPath := filepath.Join(bundle, "index.json")
|
||||||
|
if raw, ok := test.indexValue.(json.RawMessage); ok {
|
||||||
|
if err := os.WriteFile(indexPath, raw, 0o644); err != nil {
|
||||||
|
t.Fatalf("WriteFile(index) error = %v", err)
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
writeJSONFile(t, indexPath, test.indexValue)
|
||||||
|
}
|
||||||
|
if test.prepare != nil {
|
||||||
|
test.prepare(t, bundle)
|
||||||
|
}
|
||||||
|
if _, err := loadIndex(bundle, indexPath); err == nil {
|
||||||
|
t.Fatal("loadIndex() error = nil, want failure")
|
||||||
|
} else if test.wantError != "" && !strings.Contains(err.Error(), test.wantError) {
|
||||||
|
t.Fatalf("loadIndex() error = %v, want fragment %q", err, test.wantError)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
bundle := createBundleSkeleton(t)
|
||||||
|
oversizedIndex := filepath.Join(bundle, "index.json")
|
||||||
|
if err := os.WriteFile(oversizedIndex, []byte(strings.Repeat("x", maxIndexBytes+1)), 0o644); err != nil {
|
||||||
|
t.Fatalf("WriteFile(oversized index) error = %v", err)
|
||||||
|
}
|
||||||
|
if _, err := loadIndex(bundle, oversizedIndex); err == nil || !strings.Contains(err.Error(), "exceeds") {
|
||||||
|
t.Fatalf("loadIndex(oversized) error = %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLoadDiagnosticSummariesValidateBoundsAndTolerateUnknownFields(t *testing.T) {
|
||||||
|
root := t.TempDir()
|
||||||
|
rejectedPath := filepath.Join(root, "rejected.json")
|
||||||
|
warningsPath := filepath.Join(root, "warnings.json")
|
||||||
|
diagnosticsPath := filepath.Join(root, "diagnostics.json")
|
||||||
|
writeJSONFile(t, rejectedPath, map[string]any{"rejected": []any{map[string]any{
|
||||||
|
"stage": "validate", "lane_id": "spells", "reason_code": "invalid", "message": "do not retain this", "future": true,
|
||||||
|
}}, "future": true})
|
||||||
|
writeJSONFile(t, warningsPath, findingEnvelope(warningsSchemaVersion, []any{findingGroup("warning", "degradation", "bounded", "normalize", 2)}, 2, false, 0))
|
||||||
|
writeJSONFile(t, diagnosticsPath, findingEnvelope(diagnosticsSchemaVersion, []any{findingGroup("advisory", "data_quality", "low_confidence", "normalize", 3)}, 4, true, 1))
|
||||||
|
rejections, err := loadRejections(rejectedPath)
|
||||||
|
if err != nil || len(rejections) != 1 || rejections[0].ReasonCode != "invalid" {
|
||||||
|
t.Fatalf("loadRejections() = %#v, %v", rejections, err)
|
||||||
|
}
|
||||||
|
warnings, err := loadWarnings(warningsPath)
|
||||||
|
if err != nil || len(warnings) != 1 || warnings[0].Category != "degradation" || warnings[0].OccurrenceCount != 2 {
|
||||||
|
t.Fatalf("loadWarnings() = %#v, %v", warnings, err)
|
||||||
|
}
|
||||||
|
diagnostics, occurrences, truncated, err := loadDiagnostics(diagnosticsPath)
|
||||||
|
if err != nil || len(diagnostics) != 1 || occurrences != 4 || !truncated || diagnostics[0].Category != "data_quality" {
|
||||||
|
t.Fatalf("loadDiagnostics() = %#v, %d, %t, %v", diagnostics, occurrences, truncated, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
for name, path := range map[string]string{"rejections": rejectedPath, "warnings": warningsPath, "diagnostics": diagnosticsPath} {
|
||||||
|
t.Run("malformed "+name, func(t *testing.T) {
|
||||||
|
if err := os.WriteFile(path, []byte("{"), 0o644); err != nil {
|
||||||
|
t.Fatalf("WriteFile() error = %v", err)
|
||||||
|
}
|
||||||
|
var err error
|
||||||
|
if name == "rejections" {
|
||||||
|
_, err = loadRejections(path)
|
||||||
|
} else if name == "warnings" {
|
||||||
|
_, err = loadWarnings(path)
|
||||||
|
} else {
|
||||||
|
_, _, _, err = loadDiagnostics(path)
|
||||||
|
}
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("summary decoder error = nil")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
oversized := filepath.Join(root, "oversized.json")
|
||||||
|
if err := os.WriteFile(oversized, []byte(strings.Repeat("x", maxSummaryBytes+1)), 0o644); err != nil {
|
||||||
|
t.Fatalf("WriteFile(oversized) error = %v", err)
|
||||||
|
}
|
||||||
|
if _, err := loadWarnings(oversized); err == nil || !strings.Contains(err.Error(), "exceeds") {
|
||||||
|
t.Fatalf("loadWarnings(oversized) error = %v", err)
|
||||||
|
}
|
||||||
|
if _, err := loadRejections(oversized); err == nil || !strings.Contains(err.Error(), "exceeds") {
|
||||||
|
t.Fatalf("loadRejections(oversized) error = %v", err)
|
||||||
|
}
|
||||||
|
if _, _, _, err := loadDiagnostics(oversized); err == nil || !strings.Contains(err.Error(), "exceeds") {
|
||||||
|
t.Fatalf("loadDiagnostics(oversized) error = %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestFakeRunnerCapturesRequestsAndHonorsContextAndError(t *testing.T) {
|
||||||
|
req := RunRequest{PipelineID: "pipeline", References: []ReferenceBinding{{Selector: "party", Path: "/references/party.json"}}}
|
||||||
|
want := RunResult{BundleRoot: "/bundle"}
|
||||||
|
fake := &FakeRunner{Result: want}
|
||||||
|
got, err := fake.Run(context.Background(), req)
|
||||||
|
if err != nil || !reflect.DeepEqual(got, want) || !reflect.DeepEqual(fake.Requests, []RunRequest{req}) {
|
||||||
|
t.Fatalf("Run() = %#v, %v; requests = %#v", got, err, fake.Requests)
|
||||||
|
}
|
||||||
|
req.References[0].Path = "/references/changed.json"
|
||||||
|
if fake.Requests[0].References[0].Path != "/references/party.json" {
|
||||||
|
t.Fatalf("fake retained aliased request references: %#v", fake.Requests[0].References)
|
||||||
|
}
|
||||||
|
|
||||||
|
wantErr := errors.New("configured failure")
|
||||||
|
fake.Err = wantErr
|
||||||
|
if _, err := fake.Run(context.Background(), req); !errors.Is(err, wantErr) {
|
||||||
|
t.Fatalf("Run(configured error) = %v", err)
|
||||||
|
}
|
||||||
|
canceled, cancel := context.WithCancel(context.Background())
|
||||||
|
cancel()
|
||||||
|
before := len(fake.Requests)
|
||||||
|
if _, err := fake.Run(canceled, req); !errors.Is(err, context.Canceled) || len(fake.Requests) != before {
|
||||||
|
t.Fatalf("Run(canceled) error = %v; requests = %d", err, len(fake.Requests))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func validRunRequest(t *testing.T) RunRequest {
|
||||||
|
t.Helper()
|
||||||
|
root := t.TempDir()
|
||||||
|
configPath := filepath.Join(root, "notarius.yml")
|
||||||
|
inputPath := filepath.Join(root, "input.json")
|
||||||
|
outputRoot := filepath.Join(root, "outputs")
|
||||||
|
workingDirectory := filepath.Join(root, "work")
|
||||||
|
diagnostics := filepath.Join(root, "diagnostics")
|
||||||
|
for _, directory := range []string{outputRoot, workingDirectory, diagnostics} {
|
||||||
|
if err := os.Mkdir(directory, 0o755); err != nil {
|
||||||
|
t.Fatalf("Mkdir(%q) error = %v", directory, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(configPath, []byte("pipelines: {}\n"), 0o644); err != nil {
|
||||||
|
t.Fatalf("WriteFile(config) error = %v", err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(inputPath, []byte("{}\n"), 0o644); err != nil {
|
||||||
|
t.Fatalf("WriteFile(input) error = %v", err)
|
||||||
|
}
|
||||||
|
return RunRequest{
|
||||||
|
Binary: "notarius", ConfigPath: configPath, PipelineID: "dnd-session", InputPath: inputPath,
|
||||||
|
OutputRoot: outputRoot, WorkingDirectory: workingDirectory,
|
||||||
|
ReceiptPath: filepath.Join(diagnostics, "receipt.json"), LogPath: filepath.Join(diagnostics, "stderr.log"),
|
||||||
|
Timeout: time.Second,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func writeValidBundleAndReceipt(t *testing.T, req RunRequest, includeUnknown bool) {
|
||||||
|
t.Helper()
|
||||||
|
bundle := filepath.Join(req.OutputRoot, "notarius-run-1")
|
||||||
|
if err := os.MkdirAll(filepath.Join(bundle, "lanes"), 0o755); err != nil {
|
||||||
|
t.Fatalf("MkdirAll(bundle) error = %v", err)
|
||||||
|
}
|
||||||
|
for path, data := range map[string]string{
|
||||||
|
"manifest.json": `{}`,
|
||||||
|
"lanes/npc.json": `{}`,
|
||||||
|
"chunk-map.json": `{}`,
|
||||||
|
"evidence-context.json": `{}`,
|
||||||
|
} {
|
||||||
|
if err := os.WriteFile(filepath.Join(bundle, filepath.FromSlash(path)), []byte(data), 0o644); err != nil {
|
||||||
|
t.Fatalf("WriteFile(%q) error = %v", path, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
rejection := map[string]any{"stage": "validate", "lane_id": "spells", "reason_code": "invalid_spell", "message": strings.Repeat("external detail", 20)}
|
||||||
|
if includeUnknown {
|
||||||
|
rejection["future"] = true
|
||||||
|
}
|
||||||
|
writeJSONFile(t, filepath.Join(bundle, "rejected.json"), map[string]any{"rejected": []any{rejection}, "future": true})
|
||||||
|
writeJSONFile(t, filepath.Join(bundle, "warnings.json"), findingEnvelope(warningsSchemaVersion, []any{findingGroup("warning", "degradation", "normalized_name", "normalize", 2)}, 2, false, 0))
|
||||||
|
writeJSONFile(t, filepath.Join(bundle, "diagnostics.json"), findingEnvelope(diagnosticsSchemaVersion, []any{findingGroup("advisory", "data_quality", "low_confidence", "normalize", 3)}, 4, true, 1))
|
||||||
|
index := validIndexValue([]any{map[string]any{
|
||||||
|
"lane_id": "npc-registry", "file": "lanes/npc.json", "media_type": "application/json",
|
||||||
|
"module_key": "dnd/npc-registry", "schema_id": "notarius.dnd.npc_registry",
|
||||||
|
"schema_name": "NPCRegistry", "schema_version": "v1", "future": true,
|
||||||
|
}})
|
||||||
|
index["chunk_map"] = map[string]any{
|
||||||
|
"artifact_kind": "chunk_map", "file": "chunk-map.json", "media_type": "application/json",
|
||||||
|
"schema_id": "notarius.chunk_map", "schema_name": "ChunkMap", "schema_version": "v1", "future": true,
|
||||||
|
}
|
||||||
|
index["evidence_context"] = map[string]any{
|
||||||
|
"artifact_kind": "evidence_context", "file": "evidence-context.json", "media_type": "application/json",
|
||||||
|
"schema_id": "notarius.evidence_context", "schema_name": "EvidenceContext", "schema_version": "v1", "future": true,
|
||||||
|
}
|
||||||
|
index["future"] = true
|
||||||
|
writeJSONFile(t, filepath.Join(bundle, "index.json"), index)
|
||||||
|
receipt := map[string]any{
|
||||||
|
"schema_version": ReceiptSchemaVersion, "run_id": "notarius-run-1", "pipeline_id": req.PipelineID,
|
||||||
|
"output_directory": bundle, "index_file": "index.json", "normalized_output_count": 1,
|
||||||
|
"rejected_output_count": 1, "warning_group_count": 1, "warning_occurrence_count": 2,
|
||||||
|
"diagnostic_group_count": 1, "diagnostic_occurrence_count": 4,
|
||||||
|
"diagnostics_truncated": true, "validation_status": "rejected",
|
||||||
|
"validation_summaries": []any{map[string]any{
|
||||||
|
"stage": "normalize", "lane_id": "npc-registry", "status": "complete",
|
||||||
|
"producer_attempt_count": 1, "terminal_action": "accepted",
|
||||||
|
}},
|
||||||
|
}
|
||||||
|
if includeUnknown {
|
||||||
|
receipt["future"] = true
|
||||||
|
}
|
||||||
|
writeJSONFile(t, req.ReceiptPath, receipt)
|
||||||
|
}
|
||||||
|
|
||||||
|
func createBundleSkeleton(t *testing.T) string {
|
||||||
|
t.Helper()
|
||||||
|
bundle := filepath.Join(t.TempDir(), "bundle")
|
||||||
|
if err := os.MkdirAll(filepath.Join(bundle, "lanes"), 0o755); err != nil {
|
||||||
|
t.Fatalf("MkdirAll(bundle) error = %v", err)
|
||||||
|
}
|
||||||
|
for _, name := range []string{"manifest.json", "rejected.json", "warnings.json", "diagnostics.json", "lanes/npc.json", "chunk-map.json"} {
|
||||||
|
if err := os.WriteFile(filepath.Join(bundle, filepath.FromSlash(name)), []byte("{}"), 0o644); err != nil {
|
||||||
|
t.Fatalf("WriteFile(%q) error = %v", name, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return bundle
|
||||||
|
}
|
||||||
|
|
||||||
|
func validIndexValue(lanes []any) map[string]any {
|
||||||
|
return map[string]any{
|
||||||
|
"manifest_file": "manifest.json", "output_files": lanes,
|
||||||
|
"rejected_file": "rejected.json", "warnings_file": "warnings.json", "diagnostics_file": "diagnostics.json",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func findingGroup(disposition, category, reasonCode, origin string, occurrences int) map[string]any {
|
||||||
|
return map[string]any{
|
||||||
|
"disposition": disposition, "category": category, "reason_code": reasonCode,
|
||||||
|
"origin": map[string]any{"stage": origin, "lane_id": "npc-registry"}, "occurrence_count": occurrences,
|
||||||
|
"samples": []any{map[string]any{"scope": "lane:npc-registry", "message": "external detail"}},
|
||||||
|
"omitted_sample_count": occurrences - 1,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func findingEnvelope(schema string, groups []any, occurrences int, truncated bool, unrepresented int) map[string]any {
|
||||||
|
value := map[string]any{
|
||||||
|
"schema_version": schema, "group_count": len(groups), "occurrence_count": occurrences,
|
||||||
|
"groups": groups,
|
||||||
|
}
|
||||||
|
if schema == diagnosticsSchemaVersion {
|
||||||
|
value["truncated"] = truncated
|
||||||
|
value["unrepresented_occurrence_count"] = unrepresented
|
||||||
|
}
|
||||||
|
return value
|
||||||
|
}
|
||||||
|
|
||||||
|
func writeJSONFile(t *testing.T, path string, value any) {
|
||||||
|
t.Helper()
|
||||||
|
data, err := json.Marshal(value)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("json.Marshal() error = %v", err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(path, data, 0o644); err != nil {
|
||||||
|
t.Fatalf("WriteFile(%q) error = %v", path, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func writeShellScript(t *testing.T, body string) string {
|
||||||
|
t.Helper()
|
||||||
|
path := filepath.Join(t.TempDir(), "notarius-helper")
|
||||||
|
if err := os.WriteFile(path, []byte(body), 0o755); err != nil {
|
||||||
|
t.Fatalf("WriteFile(script) error = %v", err)
|
||||||
|
}
|
||||||
|
return path
|
||||||
|
}
|
||||||
|
|
||||||
|
func assertTextFile(t *testing.T, path, want string) {
|
||||||
|
t.Helper()
|
||||||
|
data, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ReadFile(%q) error = %v", path, err)
|
||||||
|
}
|
||||||
|
if string(data) != want {
|
||||||
|
t.Fatalf("ReadFile(%q) = %q, want %q", path, string(data), want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func cloneMap(source map[string]any) map[string]any {
|
||||||
|
result := make(map[string]any, len(source))
|
||||||
|
for key, value := range source {
|
||||||
|
result[key] = value
|
||||||
|
}
|
||||||
|
return result
|
||||||
|
}
|
||||||
@@ -5,6 +5,7 @@ import (
|
|||||||
"fmt"
|
"fmt"
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||||
|
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||||
)
|
)
|
||||||
|
|
||||||
// NoopRunner is a deterministic no-op scriptorium adapter.
|
// NoopRunner is a deterministic no-op scriptorium adapter.
|
||||||
@@ -142,7 +143,7 @@ func (f *FakeRunner) RenderArtifact(ctx context.Context, req RenderArtifactReque
|
|||||||
|
|
||||||
func materializeRunPlaceholders(req RunArtifactRequest) error {
|
func materializeRunPlaceholders(req RunArtifactRequest) error {
|
||||||
if req.OutputPath != "" {
|
if req.OutputPath != "" {
|
||||||
if err := subprocess.WriteFileAtomic(req.OutputPath, []byte("scriptorium noop/fake run artifact\n"), 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(req.OutputPath, []byte("scriptorium noop/fake run artifact\n"), fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write run output %q: %w", req.OutputPath, err)
|
return fmt.Errorf("write run output %q: %w", req.OutputPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -154,17 +155,17 @@ func materializeRunPlaceholders(req RunArtifactRequest) error {
|
|||||||
"prompt_id": req.PromptID,
|
"prompt_id": req.PromptID,
|
||||||
"output_path": req.OutputPath,
|
"output_path": req.OutputPath,
|
||||||
}
|
}
|
||||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if req.StdoutLogPath != "" {
|
if req.StdoutLogPath != "" {
|
||||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("scriptorium noop/fake run stdout placeholder\n"), 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("scriptorium noop/fake run stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if req.StderrLogPath != "" {
|
if req.StderrLogPath != "" {
|
||||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("scriptorium noop/fake run stderr placeholder\n"), 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("scriptorium noop/fake run stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -173,7 +174,7 @@ func materializeRunPlaceholders(req RunArtifactRequest) error {
|
|||||||
|
|
||||||
func materializeRenderPlaceholders(req RenderArtifactRequest) error {
|
func materializeRenderPlaceholders(req RenderArtifactRequest) error {
|
||||||
if req.OutputPath != "" {
|
if req.OutputPath != "" {
|
||||||
if err := subprocess.WriteFileAtomic(req.OutputPath, []byte("{\"schema\":\"scriptorium.render.v1\",\"placeholder\":true}\n"), 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(req.OutputPath, []byte("{\"schema\":\"scriptorium.render.v1\",\"placeholder\":true}\n"), fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write render output %q: %w", req.OutputPath, err)
|
return fmt.Errorf("write render output %q: %w", req.OutputPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -185,17 +186,17 @@ func materializeRenderPlaceholders(req RenderArtifactRequest) error {
|
|||||||
"prompt_id": req.PromptID,
|
"prompt_id": req.PromptID,
|
||||||
"output_path": req.OutputPath,
|
"output_path": req.OutputPath,
|
||||||
}
|
}
|
||||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if req.StdoutLogPath != "" {
|
if req.StdoutLogPath != "" {
|
||||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("scriptorium noop/fake render stdout placeholder\n"), 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("scriptorium noop/fake render stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if req.StderrLogPath != "" {
|
if req.StderrLogPath != "" {
|
||||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("scriptorium noop/fake render stderr placeholder\n"), 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("scriptorium noop/fake render stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -9,8 +9,12 @@ import (
|
|||||||
"time"
|
"time"
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||||
|
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// MaxOutputFileBytes bounds one Scriptorium artifact result.
|
||||||
|
const MaxOutputFileBytes int64 = 64 * 1024 * 1024
|
||||||
|
|
||||||
// SubprocessRunner invokes Scriptorium through its public CLI.
|
// SubprocessRunner invokes Scriptorium through its public CLI.
|
||||||
type SubprocessRunner struct{}
|
type SubprocessRunner struct{}
|
||||||
|
|
||||||
@@ -52,13 +56,17 @@ func (r *SubprocessRunner) RunArtifact(ctx context.Context, req RunArtifactReque
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
envOverrides, sensitiveNames := credentialEnvironment(req.APIKeyEnv)
|
||||||
runRes, runErr := subprocess.Run(ctx, subprocess.RunRequest{
|
runRes, runErr := subprocess.Run(ctx, subprocess.RunRequest{
|
||||||
Executable: req.Binary,
|
Executable: req.Binary,
|
||||||
Args: args,
|
Args: args,
|
||||||
WorkingDir: req.WorkingDir,
|
WorkingDir: req.WorkingDir,
|
||||||
Timeout: req.Timeout,
|
Timeout: req.Timeout,
|
||||||
StdoutLogPath: req.StdoutLogPath,
|
EnvOverrides: envOverrides,
|
||||||
StderrLogPath: req.StderrLogPath,
|
SensitiveEnvNames: sensitiveNames,
|
||||||
|
DiagnosticOwner: "scriptorium",
|
||||||
|
StdoutLogPath: req.StdoutLogPath,
|
||||||
|
StderrLogPath: req.StderrLogPath,
|
||||||
})
|
})
|
||||||
|
|
||||||
result := ArtifactResult{
|
result := ArtifactResult{
|
||||||
@@ -134,13 +142,17 @@ func (r *SubprocessRunner) RenderArtifact(ctx context.Context, req RenderArtifac
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
envOverrides, sensitiveNames := credentialEnvironment(req.APIKeyEnv)
|
||||||
runRes, runErr := subprocess.Run(ctx, subprocess.RunRequest{
|
runRes, runErr := subprocess.Run(ctx, subprocess.RunRequest{
|
||||||
Executable: req.Binary,
|
Executable: req.Binary,
|
||||||
Args: args,
|
Args: args,
|
||||||
WorkingDir: req.WorkingDir,
|
WorkingDir: req.WorkingDir,
|
||||||
Timeout: req.Timeout,
|
Timeout: req.Timeout,
|
||||||
StdoutLogPath: req.StdoutLogPath,
|
EnvOverrides: envOverrides,
|
||||||
StderrLogPath: req.StderrLogPath,
|
SensitiveEnvNames: sensitiveNames,
|
||||||
|
DiagnosticOwner: "scriptorium",
|
||||||
|
StdoutLogPath: req.StdoutLogPath,
|
||||||
|
StderrLogPath: req.StderrLogPath,
|
||||||
})
|
})
|
||||||
|
|
||||||
result := ArtifactResult{
|
result := ArtifactResult{
|
||||||
@@ -223,6 +235,15 @@ func validateCommonRunRequest(
|
|||||||
return true, nil
|
return true, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func credentialEnvironment(apiKeyEnv string) (map[string]string, []string) {
|
||||||
|
name := strings.TrimSpace(apiKeyEnv)
|
||||||
|
if name == "" {
|
||||||
|
return nil, nil
|
||||||
|
}
|
||||||
|
value, _ := os.LookupEnv(name)
|
||||||
|
return map[string]string{name: value}, []string{name}
|
||||||
|
}
|
||||||
|
|
||||||
func buildRunArgs(req RunArtifactRequest) []string {
|
func buildRunArgs(req RunArtifactRequest) []string {
|
||||||
args := []string{"run", "--prompt", strings.TrimSpace(req.PromptID)}
|
args := []string{"run", "--prompt", strings.TrimSpace(req.PromptID)}
|
||||||
if cfgPath := strings.TrimSpace(req.ConfigPath); cfgPath != "" {
|
if cfgPath := strings.TrimSpace(req.ConfigPath); cfgPath != "" {
|
||||||
@@ -321,18 +342,15 @@ func writeInvocationConfig(path string, payload invocationPayload) error {
|
|||||||
"render_format": payload.RenderFormat,
|
"render_format": payload.RenderFormat,
|
||||||
"render_prompt_logged": payload.RenderPromptStore,
|
"render_prompt_logged": payload.RenderPromptStore,
|
||||||
}
|
}
|
||||||
return subprocess.WriteYAMLAtomic(path, data, 0o644)
|
return subprocess.WriteYAMLAtomic(path, data, fileops.WorkspaceFileMode)
|
||||||
}
|
}
|
||||||
|
|
||||||
func validateNonEmptyOutput(path string) error {
|
func validateNonEmptyOutput(path string) error {
|
||||||
info, err := os.Stat(path)
|
data, err := fileops.ReadRegularFile(path, MaxOutputFileBytes)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("stat file: %w", err)
|
return fmt.Errorf("scriptorium artifact output exceeds or cannot be read within %d-byte limit: %w", MaxOutputFileBytes, err)
|
||||||
}
|
}
|
||||||
if info.IsDir() {
|
if len(data) == 0 {
|
||||||
return fmt.Errorf("path is a directory")
|
|
||||||
}
|
|
||||||
if info.Size() <= 0 {
|
|
||||||
return fmt.Errorf("file is empty")
|
return fmt.Errorf("file is empty")
|
||||||
}
|
}
|
||||||
return nil
|
return nil
|
||||||
|
|||||||
@@ -31,7 +31,7 @@ func TestSubprocessRunnerRunSuccessBuildsDeterministicArgsAndCapturesLogs(t *tes
|
|||||||
ConfigPath: "/etc/scriptorium/config.yml",
|
ConfigPath: "/etc/scriptorium/config.yml",
|
||||||
PromptID: "dnd.session_recap",
|
PromptID: "dnd.session_recap",
|
||||||
ProfileID: "local-quality",
|
ProfileID: "local-quality",
|
||||||
InputPaths: map[string]string{"transcript": filepath.Join(dir, "processed.json"), "other": filepath.Join(dir, "other.md")},
|
InputPaths: map[string]string{"transcript": filepath.Join(dir, "polished.json"), "other": filepath.Join(dir, "other.md")},
|
||||||
Vars: map[string]string{"session_id": "2026-05-03", "campaign_name": "Icewind Dale"},
|
Vars: map[string]string{"session_id": "2026-05-03", "campaign_name": "Icewind Dale"},
|
||||||
OutputPath: filepath.Join(dir, "artifacts", "session_recap.md"),
|
OutputPath: filepath.Join(dir, "artifacts", "session_recap.md"),
|
||||||
StdoutLogPath: filepath.Join(dir, "logs", "scriptorium.run.stdout.log"),
|
StdoutLogPath: filepath.Join(dir, "logs", "scriptorium.run.stdout.log"),
|
||||||
@@ -180,7 +180,7 @@ func TestSubprocessRunnerRenderSuccess(t *testing.T) {
|
|||||||
req := RenderArtifactRequest{
|
req := RenderArtifactRequest{
|
||||||
Binary: wrapper,
|
Binary: wrapper,
|
||||||
PromptID: "dnd.session_recap",
|
PromptID: "dnd.session_recap",
|
||||||
InputPaths: map[string]string{"transcript": filepath.Join(dir, "processed.json")},
|
InputPaths: map[string]string{"transcript": filepath.Join(dir, "polished.json")},
|
||||||
OutputPath: filepath.Join(dir, "artifacts", "session_recap.render.json"),
|
OutputPath: filepath.Join(dir, "artifacts", "session_recap.render.json"),
|
||||||
StdoutLogPath: filepath.Join(dir, "logs", "scriptorium.render.stdout.log"),
|
StdoutLogPath: filepath.Join(dir, "logs", "scriptorium.render.stdout.log"),
|
||||||
StderrLogPath: filepath.Join(dir, "logs", "scriptorium.render.stderr.log"),
|
StderrLogPath: filepath.Join(dir, "logs", "scriptorium.render.stderr.log"),
|
||||||
@@ -285,7 +285,7 @@ type scriptoriumHelperRecord struct {
|
|||||||
func runReqForTest(t *testing.T, binary string) RunArtifactRequest {
|
func runReqForTest(t *testing.T, binary string) RunArtifactRequest {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
dir := t.TempDir()
|
dir := t.TempDir()
|
||||||
transcriptPath := filepath.Join(dir, "processed.json")
|
transcriptPath := filepath.Join(dir, "polished.json")
|
||||||
writeScriptoriumFile(t, transcriptPath, `{"segments":[]}`)
|
writeScriptoriumFile(t, transcriptPath, `{"segments":[]}`)
|
||||||
return RunArtifactRequest{
|
return RunArtifactRequest{
|
||||||
Binary: binary,
|
Binary: binary,
|
||||||
|
|||||||
@@ -5,6 +5,7 @@ import (
|
|||||||
"fmt"
|
"fmt"
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||||
|
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||||
)
|
)
|
||||||
|
|
||||||
// NoopRunner is a deterministic no-op seriatim adapter.
|
// NoopRunner is a deterministic no-op seriatim adapter.
|
||||||
@@ -68,6 +69,26 @@ func (n *NoopRunner) Normalize(ctx context.Context, req NormalizeRequest) (Norma
|
|||||||
}, nil
|
}, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Render returns the requested output path with placeholder metadata.
|
||||||
|
func (n *NoopRunner) Render(ctx context.Context, req RenderRequest) (RenderResult, error) {
|
||||||
|
if err := ctx.Err(); err != nil {
|
||||||
|
return RenderResult{}, err
|
||||||
|
}
|
||||||
|
if err := materializeRenderPlaceholders(req); err != nil {
|
||||||
|
return RenderResult{}, err
|
||||||
|
}
|
||||||
|
return RenderResult{
|
||||||
|
OutputRenderedPath: req.OutputRenderedPath,
|
||||||
|
StdoutLogPath: req.StdoutLogPath,
|
||||||
|
StderrLogPath: req.StderrLogPath,
|
||||||
|
GeneratedConfigPath: req.GeneratedConfigPath,
|
||||||
|
InvokedBinary: "noop",
|
||||||
|
Format: req.Format,
|
||||||
|
Title: req.Title,
|
||||||
|
Metadata: map[string]any{"placeholder": true},
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
// FakeRunner captures merge requests and returns deterministic responses.
|
// FakeRunner captures merge requests and returns deterministic responses.
|
||||||
type FakeRunner struct {
|
type FakeRunner struct {
|
||||||
Requests []MergeRequest
|
Requests []MergeRequest
|
||||||
@@ -79,6 +100,9 @@ type FakeRunner struct {
|
|||||||
TrimRequests []TrimRequest
|
TrimRequests []TrimRequest
|
||||||
TrimErr error
|
TrimErr error
|
||||||
TrimResult TrimResult
|
TrimResult TrimResult
|
||||||
|
RenderRequests []RenderRequest
|
||||||
|
RenderErr error
|
||||||
|
RenderResult RenderResult
|
||||||
}
|
}
|
||||||
|
|
||||||
// Run records request and returns configured response.
|
// Run records request and returns configured response.
|
||||||
@@ -195,9 +219,49 @@ func (f *FakeRunner) Normalize(ctx context.Context, req NormalizeRequest) (Norma
|
|||||||
return res, nil
|
return res, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Render records request and returns configured response.
|
||||||
|
func (f *FakeRunner) Render(ctx context.Context, req RenderRequest) (RenderResult, error) {
|
||||||
|
if err := ctx.Err(); err != nil {
|
||||||
|
return RenderResult{}, err
|
||||||
|
}
|
||||||
|
f.RenderRequests = append(f.RenderRequests, req)
|
||||||
|
if f.RenderErr != nil {
|
||||||
|
return RenderResult{}, f.RenderErr
|
||||||
|
}
|
||||||
|
if err := materializeRenderPlaceholders(req); err != nil {
|
||||||
|
return RenderResult{}, err
|
||||||
|
}
|
||||||
|
res := f.RenderResult
|
||||||
|
if res.OutputRenderedPath == "" {
|
||||||
|
res.OutputRenderedPath = req.OutputRenderedPath
|
||||||
|
}
|
||||||
|
if res.StdoutLogPath == "" {
|
||||||
|
res.StdoutLogPath = req.StdoutLogPath
|
||||||
|
}
|
||||||
|
if res.StderrLogPath == "" {
|
||||||
|
res.StderrLogPath = req.StderrLogPath
|
||||||
|
}
|
||||||
|
if res.GeneratedConfigPath == "" {
|
||||||
|
res.GeneratedConfigPath = req.GeneratedConfigPath
|
||||||
|
}
|
||||||
|
if res.InvokedBinary == "" {
|
||||||
|
res.InvokedBinary = "fake"
|
||||||
|
}
|
||||||
|
if res.Format == "" {
|
||||||
|
res.Format = req.Format
|
||||||
|
}
|
||||||
|
if res.Title == "" {
|
||||||
|
res.Title = req.Title
|
||||||
|
}
|
||||||
|
if res.Metadata == nil {
|
||||||
|
res.Metadata = map[string]any{"fake": true}
|
||||||
|
}
|
||||||
|
return res, nil
|
||||||
|
}
|
||||||
|
|
||||||
func materializePlaceholders(req MergeRequest) error {
|
func materializePlaceholders(req MergeRequest) error {
|
||||||
if req.OutputMergedTranscriptPath != "" {
|
if req.OutputMergedTranscriptPath != "" {
|
||||||
if err := subprocess.WriteFileAtomic(req.OutputMergedTranscriptPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(req.OutputMergedTranscriptPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write merged transcript %q: %w", req.OutputMergedTranscriptPath, err)
|
return fmt.Errorf("write merged transcript %q: %w", req.OutputMergedTranscriptPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -208,22 +272,22 @@ func materializePlaceholders(req MergeRequest) error {
|
|||||||
"input_transcript_paths": req.InputTranscriptPaths,
|
"input_transcript_paths": req.InputTranscriptPaths,
|
||||||
"output_path": req.OutputMergedTranscriptPath,
|
"output_path": req.OutputMergedTranscriptPath,
|
||||||
}
|
}
|
||||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if req.StdoutLogPath != "" {
|
if req.StdoutLogPath != "" {
|
||||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake stdout placeholder\n"), 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if req.StderrLogPath != "" {
|
if req.StderrLogPath != "" {
|
||||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake stderr placeholder\n"), 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if req.ReportPath != "" {
|
if req.ReportPath != "" {
|
||||||
if err := subprocess.WriteFileAtomic(req.ReportPath, []byte(`{"schema":"seriatim.report.v1","placeholder":true}`), 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(req.ReportPath, []byte(`{"schema":"seriatim.report.v1","placeholder":true}`), fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write report %q: %w", req.ReportPath, err)
|
return fmt.Errorf("write report %q: %w", req.ReportPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -232,7 +296,7 @@ func materializePlaceholders(req MergeRequest) error {
|
|||||||
|
|
||||||
func materializeTrimPlaceholders(req TrimRequest) error {
|
func materializeTrimPlaceholders(req TrimRequest) error {
|
||||||
if req.OutputTrimmedPath != "" {
|
if req.OutputTrimmedPath != "" {
|
||||||
if err := subprocess.WriteFileAtomic(req.OutputTrimmedPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(req.OutputTrimmedPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write trimmed transcript %q: %w", req.OutputTrimmedPath, err)
|
return fmt.Errorf("write trimmed transcript %q: %w", req.OutputTrimmedPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -245,17 +309,17 @@ func materializeTrimPlaceholders(req TrimRequest) error {
|
|||||||
"output_path": req.OutputTrimmedPath,
|
"output_path": req.OutputTrimmedPath,
|
||||||
"keep_selector": req.KeepSelector,
|
"keep_selector": req.KeepSelector,
|
||||||
}
|
}
|
||||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if req.StdoutLogPath != "" {
|
if req.StdoutLogPath != "" {
|
||||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake trim stdout placeholder\n"), 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake trim stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if req.StderrLogPath != "" {
|
if req.StderrLogPath != "" {
|
||||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake trim stderr placeholder\n"), 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake trim stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -264,7 +328,7 @@ func materializeTrimPlaceholders(req TrimRequest) error {
|
|||||||
|
|
||||||
func materializeNormalizePlaceholders(req NormalizeRequest) error {
|
func materializeNormalizePlaceholders(req NormalizeRequest) error {
|
||||||
if req.OutputNormalizedPath != "" {
|
if req.OutputNormalizedPath != "" {
|
||||||
if err := subprocess.WriteFileAtomic(req.OutputNormalizedPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(req.OutputNormalizedPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write normalized transcript %q: %w", req.OutputNormalizedPath, err)
|
return fmt.Errorf("write normalized transcript %q: %w", req.OutputNormalizedPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -280,24 +344,60 @@ func materializeNormalizePlaceholders(req NormalizeRequest) error {
|
|||||||
if req.ReportPath != "" {
|
if req.ReportPath != "" {
|
||||||
payload["report_path"] = req.ReportPath
|
payload["report_path"] = req.ReportPath
|
||||||
}
|
}
|
||||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if req.StdoutLogPath != "" {
|
if req.StdoutLogPath != "" {
|
||||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake normalize stdout placeholder\n"), 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake normalize stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if req.StderrLogPath != "" {
|
if req.StderrLogPath != "" {
|
||||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake normalize stderr placeholder\n"), 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake normalize stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if req.ReportPath != "" {
|
if req.ReportPath != "" {
|
||||||
if err := subprocess.WriteFileAtomic(req.ReportPath, []byte(`{"schema":"seriatim.report.v1","placeholder":true}`), 0o644); err != nil {
|
if err := subprocess.WriteFileAtomic(req.ReportPath, []byte(`{"schema":"seriatim.report.v1","placeholder":true}`), fileops.WorkspaceFileMode); err != nil {
|
||||||
return fmt.Errorf("write report %q: %w", req.ReportPath, err)
|
return fmt.Errorf("write report %q: %w", req.ReportPath, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func materializeRenderPlaceholders(req RenderRequest) error {
|
||||||
|
if req.OutputRenderedPath != "" {
|
||||||
|
if err := subprocess.WriteFileAtomic(req.OutputRenderedPath, []byte("# Transcript\n\nRendered markdown placeholder.\n"), fileops.WorkspaceFileMode); err != nil {
|
||||||
|
return fmt.Errorf("write rendered transcript %q: %w", req.OutputRenderedPath, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if req.GeneratedConfigPath != "" {
|
||||||
|
payload := map[string]any{
|
||||||
|
"schema": "seriatim.generated.v1",
|
||||||
|
"placeholder": true,
|
||||||
|
"command": "render",
|
||||||
|
"input_path": req.InputTranscriptPath,
|
||||||
|
"output_path": req.OutputRenderedPath,
|
||||||
|
"format": req.Format,
|
||||||
|
"title": req.Title,
|
||||||
|
"include_timestamps": req.IncludeTimestamps,
|
||||||
|
"include_segment_ids": req.IncludeSegmentIDs,
|
||||||
|
"include_metadata": req.IncludeMetadata,
|
||||||
|
}
|
||||||
|
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||||
|
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if req.StdoutLogPath != "" {
|
||||||
|
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake render stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||||
|
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if req.StderrLogPath != "" {
|
||||||
|
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake render stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||||
|
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|||||||
@@ -14,7 +14,7 @@ func TestFakeRunnerCapturesRequestAndReturnsPath(t *testing.T) {
|
|||||||
dir := t.TempDir()
|
dir := t.TempDir()
|
||||||
req := MergeRequest{
|
req := MergeRequest{
|
||||||
GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.yml"),
|
GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.yml"),
|
||||||
OutputMergedTranscriptPath: filepath.Join(dir, "transcripts", "merged.json"),
|
OutputMergedTranscriptPath: filepath.Join(dir, "transcripts", "base.json"),
|
||||||
StdoutLogPath: filepath.Join(dir, "logs", "seriatim.stdout.log"),
|
StdoutLogPath: filepath.Join(dir, "logs", "seriatim.stdout.log"),
|
||||||
StderrLogPath: filepath.Join(dir, "logs", "seriatim.stderr.log"),
|
StderrLogPath: filepath.Join(dir, "logs", "seriatim.stderr.log"),
|
||||||
}
|
}
|
||||||
@@ -57,8 +57,8 @@ func TestFakeRunnerTrimCapturesRequestAndReturnsPath(t *testing.T) {
|
|||||||
dir := t.TempDir()
|
dir := t.TempDir()
|
||||||
req := TrimRequest{
|
req := TrimRequest{
|
||||||
GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.trim.yml"),
|
GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.trim.yml"),
|
||||||
InputTranscriptPath: filepath.Join(dir, "transcripts", "processed.json"),
|
InputTranscriptPath: filepath.Join(dir, "transcripts", "polished.json"),
|
||||||
OutputTrimmedPath: filepath.Join(dir, "transcripts", "trimmed.json"),
|
OutputTrimmedPath: filepath.Join(dir, "transcripts", "final.trimmed.json"),
|
||||||
KeepSelector: "1-10",
|
KeepSelector: "1-10",
|
||||||
StdoutLogPath: filepath.Join(dir, "logs", "seriatim.trim.stdout.log"),
|
StdoutLogPath: filepath.Join(dir, "logs", "seriatim.trim.stdout.log"),
|
||||||
StderrLogPath: filepath.Join(dir, "logs", "seriatim.trim.stderr.log"),
|
StderrLogPath: filepath.Join(dir, "logs", "seriatim.trim.stderr.log"),
|
||||||
@@ -105,8 +105,8 @@ func TestFakeRunnerNormalizeCapturesRequestAndReturnsPath(t *testing.T) {
|
|||||||
dir := t.TempDir()
|
dir := t.TempDir()
|
||||||
req := NormalizeRequest{
|
req := NormalizeRequest{
|
||||||
GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.normalize.yml"),
|
GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.normalize.yml"),
|
||||||
InputTranscriptPath: filepath.Join(dir, "transcripts", "processed.json"),
|
InputTranscriptPath: filepath.Join(dir, "transcripts", "polished.json"),
|
||||||
OutputNormalizedPath: filepath.Join(dir, "transcripts", "normalized.json"),
|
OutputNormalizedPath: filepath.Join(dir, "transcripts", "final.json"),
|
||||||
OutputSchema: "seriatim-intermediate",
|
OutputSchema: "seriatim-intermediate",
|
||||||
ReportPath: filepath.Join(dir, "artifacts", "seriatim.normalize.report.json"),
|
ReportPath: filepath.Join(dir, "artifacts", "seriatim.normalize.report.json"),
|
||||||
StdoutLogPath: filepath.Join(dir, "logs", "seriatim.normalize.stdout.log"),
|
StdoutLogPath: filepath.Join(dir, "logs", "seriatim.normalize.stdout.log"),
|
||||||
@@ -148,3 +148,58 @@ func TestFakeRunnerNormalizeError(t *testing.T) {
|
|||||||
t.Fatal("expected error, got nil")
|
t.Fatal("expected error, got nil")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestFakeRunnerRenderCapturesRequestAndReturnsPath(t *testing.T) {
|
||||||
|
fake := &FakeRunner{}
|
||||||
|
dir := t.TempDir()
|
||||||
|
req := RenderRequest{
|
||||||
|
GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.render.yml"),
|
||||||
|
InputTranscriptPath: filepath.Join(dir, "transcripts", "final.trimmed.json"),
|
||||||
|
OutputRenderedPath: filepath.Join(dir, "transcripts", "final.trimmed.md"),
|
||||||
|
Format: "markdown",
|
||||||
|
Title: "Session render",
|
||||||
|
IncludeTimestamps: true,
|
||||||
|
IncludeSegmentIDs: false,
|
||||||
|
IncludeMetadata: true,
|
||||||
|
StdoutLogPath: filepath.Join(dir, "logs", "seriatim.render.stdout.log"),
|
||||||
|
StderrLogPath: filepath.Join(dir, "logs", "seriatim.render.stderr.log"),
|
||||||
|
}
|
||||||
|
|
||||||
|
res, err := fake.Render(context.Background(), req)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Render() error = %v", err)
|
||||||
|
}
|
||||||
|
if len(fake.RenderRequests) != 1 || fake.RenderRequests[0].GeneratedConfigPath == "" {
|
||||||
|
t.Fatalf("render requests = %#v, want captured request", fake.RenderRequests)
|
||||||
|
}
|
||||||
|
if res.OutputRenderedPath != req.OutputRenderedPath {
|
||||||
|
t.Fatalf("rendered path = %q, want %q", res.OutputRenderedPath, req.OutputRenderedPath)
|
||||||
|
}
|
||||||
|
if res.Format != req.Format {
|
||||||
|
t.Fatalf("format = %q, want %q", res.Format, req.Format)
|
||||||
|
}
|
||||||
|
if res.Title != req.Title {
|
||||||
|
t.Fatalf("title = %q, want %q", res.Title, req.Title)
|
||||||
|
}
|
||||||
|
|
||||||
|
cfgData, err := os.ReadFile(req.GeneratedConfigPath)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read generated config: %v", err)
|
||||||
|
}
|
||||||
|
if !strings.Contains(string(cfgData), "command: render") {
|
||||||
|
t.Fatalf("generated config = %q, want render command marker", string(cfgData))
|
||||||
|
}
|
||||||
|
for _, path := range []string{req.StdoutLogPath, req.StderrLogPath, req.OutputRenderedPath} {
|
||||||
|
if _, err := os.Stat(path); err != nil {
|
||||||
|
t.Fatalf("expected file %q to exist: %v", path, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestFakeRunnerRenderError(t *testing.T) {
|
||||||
|
fake := &FakeRunner{RenderErr: errors.New("boom")}
|
||||||
|
_, err := fake.Render(context.Background(), RenderRequest{})
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("expected error, got nil")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
// Package seriatim declares the adapter contract for transcript merge/normalize/trim execution.
|
// Package seriatim declares the adapter contract for transcript merge/normalize/trim/render execution.
|
||||||
package seriatim
|
package seriatim
|
||||||
|
|
||||||
import (
|
import (
|
||||||
@@ -6,11 +6,12 @@ import (
|
|||||||
"time"
|
"time"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Runner is the adapter boundary for seriatim merge/normalize/trim invocations.
|
// Runner is the adapter boundary for seriatim merge/normalize/trim/render invocations.
|
||||||
type Runner interface {
|
type Runner interface {
|
||||||
Run(ctx context.Context, req MergeRequest) (MergeResult, error)
|
Run(ctx context.Context, req MergeRequest) (MergeResult, error)
|
||||||
Normalize(ctx context.Context, req NormalizeRequest) (NormalizeResult, error)
|
Normalize(ctx context.Context, req NormalizeRequest) (NormalizeResult, error)
|
||||||
Trim(ctx context.Context, req TrimRequest) (TrimResult, error)
|
Trim(ctx context.Context, req TrimRequest) (TrimResult, error)
|
||||||
|
Render(ctx context.Context, req RenderRequest) (RenderResult, error)
|
||||||
}
|
}
|
||||||
|
|
||||||
// MergeRequest describes a seriatim merge invocation.
|
// MergeRequest describes a seriatim merge invocation.
|
||||||
@@ -90,3 +91,33 @@ type TrimResult struct {
|
|||||||
KeepSelector string
|
KeepSelector string
|
||||||
Metadata map[string]any
|
Metadata map[string]any
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// RenderRequest describes a seriatim render invocation.
|
||||||
|
type RenderRequest struct {
|
||||||
|
Binary string
|
||||||
|
InputTranscriptPath string
|
||||||
|
OutputRenderedPath string
|
||||||
|
Format string
|
||||||
|
Title string
|
||||||
|
IncludeTimestamps bool
|
||||||
|
IncludeSegmentIDs bool
|
||||||
|
IncludeMetadata bool
|
||||||
|
StdoutLogPath string
|
||||||
|
StderrLogPath string
|
||||||
|
GeneratedConfigPath string
|
||||||
|
Timeout time.Duration
|
||||||
|
}
|
||||||
|
|
||||||
|
// RenderResult describes a render output.
|
||||||
|
type RenderResult struct {
|
||||||
|
OutputRenderedPath string
|
||||||
|
StdoutLogPath string
|
||||||
|
StderrLogPath string
|
||||||
|
GeneratedConfigPath string
|
||||||
|
ExitCode int
|
||||||
|
Duration time.Duration
|
||||||
|
InvokedBinary string
|
||||||
|
Format string
|
||||||
|
Title string
|
||||||
|
Metadata map[string]any
|
||||||
|
}
|
||||||
|
|||||||
@@ -4,14 +4,18 @@ import (
|
|||||||
"context"
|
"context"
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
"fmt"
|
"fmt"
|
||||||
"os"
|
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
"time"
|
"time"
|
||||||
|
"unicode/utf8"
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||||
|
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// MaxOutputFileBytes bounds each Seriatim JSON or rendered-text result.
|
||||||
|
const MaxOutputFileBytes int64 = 64 * 1024 * 1024
|
||||||
|
|
||||||
// EnvConfig defines optional Seriatim environment tuning values.
|
// EnvConfig defines optional Seriatim environment tuning values.
|
||||||
type EnvConfig struct {
|
type EnvConfig struct {
|
||||||
OverlapWordRunGap *float64
|
OverlapWordRunGap *float64
|
||||||
@@ -128,12 +132,13 @@ func (r *SubprocessRunner) Run(ctx context.Context, req MergeRequest) (MergeResu
|
|||||||
}
|
}
|
||||||
|
|
||||||
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
|
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
|
||||||
Executable: r.binary,
|
Executable: r.binary,
|
||||||
Args: args,
|
Args: args,
|
||||||
Timeout: r.timeout,
|
Timeout: r.timeout,
|
||||||
EnvOverrides: env,
|
EnvOverrides: env,
|
||||||
StdoutLogPath: req.StdoutLogPath,
|
DiagnosticOwner: "seriatim",
|
||||||
StderrLogPath: req.StderrLogPath,
|
StdoutLogPath: req.StdoutLogPath,
|
||||||
|
StderrLogPath: req.StderrLogPath,
|
||||||
})
|
})
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return MergeResult{
|
return MergeResult{
|
||||||
@@ -231,11 +236,12 @@ func (r *SubprocessRunner) Trim(ctx context.Context, req TrimRequest) (TrimResul
|
|||||||
}
|
}
|
||||||
|
|
||||||
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
|
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
|
||||||
Executable: binary,
|
Executable: binary,
|
||||||
Args: args,
|
Args: args,
|
||||||
Timeout: timeout,
|
Timeout: timeout,
|
||||||
StdoutLogPath: req.StdoutLogPath,
|
DiagnosticOwner: "seriatim",
|
||||||
StderrLogPath: req.StderrLogPath,
|
StdoutLogPath: req.StdoutLogPath,
|
||||||
|
StderrLogPath: req.StderrLogPath,
|
||||||
})
|
})
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return TrimResult{
|
return TrimResult{
|
||||||
@@ -319,11 +325,12 @@ func (r *SubprocessRunner) Normalize(ctx context.Context, req NormalizeRequest)
|
|||||||
}
|
}
|
||||||
|
|
||||||
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
|
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
|
||||||
Executable: binary,
|
Executable: binary,
|
||||||
Args: args,
|
Args: args,
|
||||||
Timeout: timeout,
|
Timeout: timeout,
|
||||||
StdoutLogPath: req.StdoutLogPath,
|
DiagnosticOwner: "seriatim",
|
||||||
StderrLogPath: req.StderrLogPath,
|
StdoutLogPath: req.StdoutLogPath,
|
||||||
|
StderrLogPath: req.StderrLogPath,
|
||||||
})
|
})
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return NormalizeResult{
|
return NormalizeResult{
|
||||||
@@ -384,6 +391,97 @@ func (r *SubprocessRunner) Normalize(ctx context.Context, req NormalizeRequest)
|
|||||||
}, nil
|
}, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Render executes Seriatim render with deterministic flags and validates non-empty text output.
|
||||||
|
func (r *SubprocessRunner) Render(ctx context.Context, req RenderRequest) (RenderResult, error) {
|
||||||
|
if r == nil {
|
||||||
|
return RenderResult{}, fmt.Errorf("seriatim subprocess runner is nil")
|
||||||
|
}
|
||||||
|
if strings.TrimSpace(req.InputTranscriptPath) == "" {
|
||||||
|
return RenderResult{}, fmt.Errorf("seriatim render input path is required")
|
||||||
|
}
|
||||||
|
if strings.TrimSpace(req.OutputRenderedPath) == "" {
|
||||||
|
return RenderResult{}, fmt.Errorf("seriatim render output path is required")
|
||||||
|
}
|
||||||
|
format := strings.TrimSpace(req.Format)
|
||||||
|
if format == "" {
|
||||||
|
format = "markdown"
|
||||||
|
}
|
||||||
|
if format != "markdown" {
|
||||||
|
return RenderResult{}, fmt.Errorf("seriatim render format %q is unsupported", req.Format)
|
||||||
|
}
|
||||||
|
|
||||||
|
binary := r.binary
|
||||||
|
if strings.TrimSpace(req.Binary) != "" {
|
||||||
|
binary = strings.TrimSpace(req.Binary)
|
||||||
|
}
|
||||||
|
|
||||||
|
timeout := r.timeout
|
||||||
|
if req.Timeout < 0 {
|
||||||
|
return RenderResult{}, fmt.Errorf("seriatim render timeout must be >= 0")
|
||||||
|
}
|
||||||
|
if req.Timeout > 0 {
|
||||||
|
timeout = req.Timeout
|
||||||
|
}
|
||||||
|
|
||||||
|
args := buildRenderArgs(req, format)
|
||||||
|
if req.GeneratedConfigPath != "" {
|
||||||
|
if err := writeRenderInvocationConfig(req, args, binary, timeout, format); err != nil {
|
||||||
|
return RenderResult{}, fmt.Errorf("write seriatim render invocation config %q: %w", req.GeneratedConfigPath, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
|
||||||
|
Executable: binary,
|
||||||
|
Args: args,
|
||||||
|
Timeout: timeout,
|
||||||
|
DiagnosticOwner: "seriatim",
|
||||||
|
StdoutLogPath: req.StdoutLogPath,
|
||||||
|
StderrLogPath: req.StderrLogPath,
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
return RenderResult{
|
||||||
|
OutputRenderedPath: req.OutputRenderedPath,
|
||||||
|
StdoutLogPath: req.StdoutLogPath,
|
||||||
|
StderrLogPath: req.StderrLogPath,
|
||||||
|
GeneratedConfigPath: req.GeneratedConfigPath,
|
||||||
|
ExitCode: runRes.ExitCode,
|
||||||
|
Duration: runRes.Duration,
|
||||||
|
InvokedBinary: binary,
|
||||||
|
Format: format,
|
||||||
|
Title: req.Title,
|
||||||
|
}, fmt.Errorf("run seriatim render (binary=%q): %w", binary, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := validateNonEmptyTextFile(req.OutputRenderedPath); err != nil {
|
||||||
|
return RenderResult{
|
||||||
|
OutputRenderedPath: req.OutputRenderedPath,
|
||||||
|
StdoutLogPath: req.StdoutLogPath,
|
||||||
|
StderrLogPath: req.StderrLogPath,
|
||||||
|
GeneratedConfigPath: req.GeneratedConfigPath,
|
||||||
|
ExitCode: runRes.ExitCode,
|
||||||
|
Duration: runRes.Duration,
|
||||||
|
InvokedBinary: binary,
|
||||||
|
Format: format,
|
||||||
|
Title: req.Title,
|
||||||
|
}, fmt.Errorf("validate seriatim rendered output %q: %w", req.OutputRenderedPath, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return RenderResult{
|
||||||
|
OutputRenderedPath: req.OutputRenderedPath,
|
||||||
|
StdoutLogPath: req.StdoutLogPath,
|
||||||
|
StderrLogPath: req.StderrLogPath,
|
||||||
|
GeneratedConfigPath: req.GeneratedConfigPath,
|
||||||
|
ExitCode: runRes.ExitCode,
|
||||||
|
Duration: runRes.Duration,
|
||||||
|
InvokedBinary: binary,
|
||||||
|
Format: format,
|
||||||
|
Title: req.Title,
|
||||||
|
Metadata: map[string]any{
|
||||||
|
"adapter": "seriatim_subprocess",
|
||||||
|
},
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
func (r *SubprocessRunner) buildMergeArgs(req MergeRequest) []string {
|
func (r *SubprocessRunner) buildMergeArgs(req MergeRequest) []string {
|
||||||
args := []string{"merge"}
|
args := []string{"merge"}
|
||||||
|
|
||||||
@@ -455,7 +553,7 @@ func (r *SubprocessRunner) writeMergeInvocationConfig(req MergeRequest, args []s
|
|||||||
payload["coalesce_gap"] = *r.coalesceGap
|
payload["coalesce_gap"] = *r.coalesceGap
|
||||||
}
|
}
|
||||||
|
|
||||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
|
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
|
||||||
}
|
}
|
||||||
|
|
||||||
func buildTrimArgs(req TrimRequest) []string {
|
func buildTrimArgs(req TrimRequest) []string {
|
||||||
@@ -480,6 +578,22 @@ func buildNormalizeArgs(req NormalizeRequest, outputSchema string) []string {
|
|||||||
return args
|
return args
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func buildRenderArgs(req RenderRequest, format string) []string {
|
||||||
|
args := []string{
|
||||||
|
"render",
|
||||||
|
"--input-file", req.InputTranscriptPath,
|
||||||
|
"--output-file", req.OutputRenderedPath,
|
||||||
|
"--format", format,
|
||||||
|
"--include-timestamps=" + strconv.FormatBool(req.IncludeTimestamps),
|
||||||
|
"--include-segment-ids=" + strconv.FormatBool(req.IncludeSegmentIDs),
|
||||||
|
"--include-metadata=" + strconv.FormatBool(req.IncludeMetadata),
|
||||||
|
}
|
||||||
|
if strings.TrimSpace(req.Title) != "" {
|
||||||
|
args = append(args, "--title", req.Title)
|
||||||
|
}
|
||||||
|
return args
|
||||||
|
}
|
||||||
|
|
||||||
func writeTrimInvocationConfig(req TrimRequest, args []string, binary string, timeout time.Duration) error {
|
func writeTrimInvocationConfig(req TrimRequest, args []string, binary string, timeout time.Duration) error {
|
||||||
payload := map[string]any{
|
payload := map[string]any{
|
||||||
"schema": "seriatim.generated.v1",
|
"schema": "seriatim.generated.v1",
|
||||||
@@ -491,7 +605,7 @@ func writeTrimInvocationConfig(req TrimRequest, args []string, binary string, ti
|
|||||||
"output_path": req.OutputTrimmedPath,
|
"output_path": req.OutputTrimmedPath,
|
||||||
"keep_selector": req.KeepSelector,
|
"keep_selector": req.KeepSelector,
|
||||||
}
|
}
|
||||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
|
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
|
||||||
}
|
}
|
||||||
|
|
||||||
func writeNormalizeInvocationConfig(req NormalizeRequest, args []string, binary string, timeout time.Duration, outputSchema string) error {
|
func writeNormalizeInvocationConfig(req NormalizeRequest, args []string, binary string, timeout time.Duration, outputSchema string) error {
|
||||||
@@ -506,13 +620,31 @@ func writeNormalizeInvocationConfig(req NormalizeRequest, args []string, binary
|
|||||||
"output_schema": outputSchema,
|
"output_schema": outputSchema,
|
||||||
"report_path": req.ReportPath,
|
"report_path": req.ReportPath,
|
||||||
}
|
}
|
||||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
|
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
|
||||||
|
}
|
||||||
|
|
||||||
|
func writeRenderInvocationConfig(req RenderRequest, args []string, binary string, timeout time.Duration, format string) error {
|
||||||
|
payload := map[string]any{
|
||||||
|
"schema": "seriatim.generated.v1",
|
||||||
|
"command": "render",
|
||||||
|
"binary": binary,
|
||||||
|
"args": args,
|
||||||
|
"timeout": timeout.String(),
|
||||||
|
"input_path": req.InputTranscriptPath,
|
||||||
|
"output_path": req.OutputRenderedPath,
|
||||||
|
"format": format,
|
||||||
|
"title": req.Title,
|
||||||
|
"include_timestamps": req.IncludeTimestamps,
|
||||||
|
"include_segment_ids": req.IncludeSegmentIDs,
|
||||||
|
"include_metadata": req.IncludeMetadata,
|
||||||
|
}
|
||||||
|
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
|
||||||
}
|
}
|
||||||
|
|
||||||
func validateJSONFile(path string) error {
|
func validateJSONFile(path string) error {
|
||||||
data, err := os.ReadFile(path)
|
data, err := readSeriatimResult(path, "JSON output")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("read file: %w", err)
|
return err
|
||||||
}
|
}
|
||||||
var v any
|
var v any
|
||||||
if err := json.Unmarshal(data, &v); err != nil {
|
if err := json.Unmarshal(data, &v); err != nil {
|
||||||
@@ -522,9 +654,9 @@ func validateJSONFile(path string) error {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func validateJSONFileWithSegments(path string) error {
|
func validateJSONFileWithSegments(path string) error {
|
||||||
data, err := os.ReadFile(path)
|
data, err := readSeriatimResult(path, "transcript JSON output")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("read file: %w", err)
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
var payload map[string]any
|
var payload map[string]any
|
||||||
@@ -541,3 +673,28 @@ func validateJSONFileWithSegments(path string) error {
|
|||||||
}
|
}
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func validateNonEmptyTextFile(path string) error {
|
||||||
|
data, err := readSeriatimResult(path, "rendered text output")
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if len(data) == 0 {
|
||||||
|
return fmt.Errorf("file is empty")
|
||||||
|
}
|
||||||
|
if !utf8.Valid(data) {
|
||||||
|
return fmt.Errorf("file is not valid utf-8 text")
|
||||||
|
}
|
||||||
|
if strings.TrimSpace(string(data)) == "" {
|
||||||
|
return fmt.Errorf("file has no non-whitespace content")
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func readSeriatimResult(path, category string) ([]byte, error) {
|
||||||
|
data, err := fileops.ReadRegularFile(path, MaxOutputFileBytes)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("seriatim %s exceeds or cannot be read within %d-byte limit: %w", category, MaxOutputFileBytes, err)
|
||||||
|
}
|
||||||
|
return data, nil
|
||||||
|
}
|
||||||
|
|||||||
@@ -50,7 +50,7 @@ func TestSubprocessRunnerSuccessWithReportArgsAndEnv(t *testing.T) {
|
|||||||
req := MergeRequest{
|
req := MergeRequest{
|
||||||
GeneratedConfigPath: filepath.Join(dir, "seriatim.generated.yml"),
|
GeneratedConfigPath: filepath.Join(dir, "seriatim.generated.yml"),
|
||||||
InputTranscriptPaths: []string{filepath.Join(dir, "a.json"), filepath.Join(dir, "b.json")},
|
InputTranscriptPaths: []string{filepath.Join(dir, "a.json"), filepath.Join(dir, "b.json")},
|
||||||
OutputMergedTranscriptPath: filepath.Join(dir, "merged.json"),
|
OutputMergedTranscriptPath: filepath.Join(dir, "base.json"),
|
||||||
ReportPath: filepath.Join(dir, "seriatim.report.json"),
|
ReportPath: filepath.Join(dir, "seriatim.report.json"),
|
||||||
SpeakersPath: filepath.Join(dir, "speakers.yml"),
|
SpeakersPath: filepath.Join(dir, "speakers.yml"),
|
||||||
AutocorrectPath: filepath.Join(dir, "autocorrect.yml"),
|
AutocorrectPath: filepath.Join(dir, "autocorrect.yml"),
|
||||||
@@ -569,6 +569,156 @@ func TestSubprocessRunnerNormalizeInvalidReportJSONFails(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestSubprocessRunnerRenderSuccessInvocationAndProvenance(t *testing.T) {
|
||||||
|
if runtime.GOOS == "windows" {
|
||||||
|
t.Skip("helper wrapper script uses /bin/sh")
|
||||||
|
}
|
||||||
|
|
||||||
|
t.Setenv("GO_WANT_SERIATIM_HELPER", "1")
|
||||||
|
t.Setenv("SERIATIM_HELPER_MODE", "render_success")
|
||||||
|
recordPath := filepath.Join(t.TempDir(), "record.json")
|
||||||
|
t.Setenv("SERIATIM_HELPER_RECORD_PATH", recordPath)
|
||||||
|
|
||||||
|
wrapper := writeHelperWrapper(t)
|
||||||
|
runner := mustRunner(t, wrapper, false)
|
||||||
|
req := renderReqForTest(t)
|
||||||
|
|
||||||
|
res, err := runner.Render(context.Background(), req)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Render() error = %v", err)
|
||||||
|
}
|
||||||
|
if res.OutputRenderedPath != req.OutputRenderedPath {
|
||||||
|
t.Fatalf("OutputRenderedPath = %q, want %q", res.OutputRenderedPath, req.OutputRenderedPath)
|
||||||
|
}
|
||||||
|
if res.Format != req.Format {
|
||||||
|
t.Fatalf("Format = %q, want %q", res.Format, req.Format)
|
||||||
|
}
|
||||||
|
if res.Title != req.Title {
|
||||||
|
t.Fatalf("Title = %q, want %q", res.Title, req.Title)
|
||||||
|
}
|
||||||
|
if res.InvokedBinary != wrapper {
|
||||||
|
t.Fatalf("InvokedBinary = %q, want %q", res.InvokedBinary, wrapper)
|
||||||
|
}
|
||||||
|
if res.ExitCode != 0 {
|
||||||
|
t.Fatalf("ExitCode = %d, want 0", res.ExitCode)
|
||||||
|
}
|
||||||
|
if res.Duration <= 0 {
|
||||||
|
t.Fatalf("Duration = %s, want >0", res.Duration)
|
||||||
|
}
|
||||||
|
if res.Metadata == nil || res.Metadata["adapter"] != "seriatim_subprocess" {
|
||||||
|
t.Fatalf("Metadata = %#v, want adapter marker", res.Metadata)
|
||||||
|
}
|
||||||
|
|
||||||
|
if _, err := os.Stat(req.OutputRenderedPath); err != nil {
|
||||||
|
t.Fatalf("rendered output missing: %v", err)
|
||||||
|
}
|
||||||
|
if _, err := os.Stat(req.StdoutLogPath); err != nil {
|
||||||
|
t.Fatalf("stdout log missing: %v", err)
|
||||||
|
}
|
||||||
|
if _, err := os.Stat(req.StderrLogPath); err != nil {
|
||||||
|
t.Fatalf("stderr log missing: %v", err)
|
||||||
|
}
|
||||||
|
if _, err := os.Stat(req.GeneratedConfigPath); err != nil {
|
||||||
|
t.Fatalf("generated config missing: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
rec := readHelperRecord(t, recordPath)
|
||||||
|
wantArgs := []string{
|
||||||
|
"render",
|
||||||
|
"--input-file", req.InputTranscriptPath,
|
||||||
|
"--output-file", req.OutputRenderedPath,
|
||||||
|
"--format", req.Format,
|
||||||
|
"--include-timestamps=true",
|
||||||
|
"--include-segment-ids=true",
|
||||||
|
"--include-metadata=false",
|
||||||
|
"--title", req.Title,
|
||||||
|
}
|
||||||
|
if strings.Join(rec.Args, "\n") != strings.Join(wantArgs, "\n") {
|
||||||
|
t.Fatalf("args = %#v, want %#v", rec.Args, wantArgs)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSubprocessRunnerRenderWithoutTitleOmitsTitleArg(t *testing.T) {
|
||||||
|
if runtime.GOOS == "windows" {
|
||||||
|
t.Skip("helper wrapper script uses /bin/sh")
|
||||||
|
}
|
||||||
|
t.Setenv("GO_WANT_SERIATIM_HELPER", "1")
|
||||||
|
t.Setenv("SERIATIM_HELPER_MODE", "render_success")
|
||||||
|
recordPath := filepath.Join(t.TempDir(), "record.json")
|
||||||
|
t.Setenv("SERIATIM_HELPER_RECORD_PATH", recordPath)
|
||||||
|
|
||||||
|
runner := mustRunner(t, writeHelperWrapper(t), false)
|
||||||
|
req := renderReqForTest(t)
|
||||||
|
req.Title = ""
|
||||||
|
if _, err := runner.Render(context.Background(), req); err != nil {
|
||||||
|
t.Fatalf("Render() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
rec := readHelperRecord(t, recordPath)
|
||||||
|
for i := 0; i < len(rec.Args); i++ {
|
||||||
|
if rec.Args[i] == "--title" {
|
||||||
|
t.Fatalf("args = %#v, did not expect --title", rec.Args)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSubprocessRunnerRenderSubprocessFailure(t *testing.T) {
|
||||||
|
if runtime.GOOS == "windows" {
|
||||||
|
t.Skip("helper wrapper script uses /bin/sh")
|
||||||
|
}
|
||||||
|
t.Setenv("GO_WANT_SERIATIM_HELPER", "1")
|
||||||
|
t.Setenv("SERIATIM_HELPER_MODE", "fail")
|
||||||
|
t.Setenv("SERIATIM_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
|
||||||
|
|
||||||
|
runner := mustRunner(t, writeHelperWrapper(t), false)
|
||||||
|
req := renderReqForTest(t)
|
||||||
|
_, err := runner.Render(context.Background(), req)
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("Render() error = nil, want non-nil")
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), "run seriatim render") {
|
||||||
|
t.Fatalf("error = %q, want subprocess context", err.Error())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSubprocessRunnerRenderMissingOutputFails(t *testing.T) {
|
||||||
|
if runtime.GOOS == "windows" {
|
||||||
|
t.Skip("helper wrapper script uses /bin/sh")
|
||||||
|
}
|
||||||
|
t.Setenv("GO_WANT_SERIATIM_HELPER", "1")
|
||||||
|
t.Setenv("SERIATIM_HELPER_MODE", "missing_output")
|
||||||
|
t.Setenv("SERIATIM_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
|
||||||
|
|
||||||
|
runner := mustRunner(t, writeHelperWrapper(t), false)
|
||||||
|
req := renderReqForTest(t)
|
||||||
|
_, err := runner.Render(context.Background(), req)
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("Render() error = nil, want non-nil")
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), "validate seriatim rendered output") {
|
||||||
|
t.Fatalf("error = %q, want output validation context", err.Error())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSubprocessRunnerRenderEmptyOutputFails(t *testing.T) {
|
||||||
|
if runtime.GOOS == "windows" {
|
||||||
|
t.Skip("helper wrapper script uses /bin/sh")
|
||||||
|
}
|
||||||
|
t.Setenv("GO_WANT_SERIATIM_HELPER", "1")
|
||||||
|
t.Setenv("SERIATIM_HELPER_MODE", "render_empty_output")
|
||||||
|
t.Setenv("SERIATIM_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
|
||||||
|
|
||||||
|
runner := mustRunner(t, writeHelperWrapper(t), false)
|
||||||
|
req := renderReqForTest(t)
|
||||||
|
_, err := runner.Render(context.Background(), req)
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("Render() error = nil, want non-nil")
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), "file is empty") {
|
||||||
|
t.Fatalf("error = %q, want empty-file validation", err.Error())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestSubprocessRunnerConstructorValidation(t *testing.T) {
|
func TestSubprocessRunnerConstructorValidation(t *testing.T) {
|
||||||
_, err := NewSubprocessRunnerFromConfigValues("", "10m", "seriatim-intermediate", nil, true, EnvConfig{})
|
_, err := NewSubprocessRunnerFromConfigValues("", "10m", "seriatim-intermediate", nil, true, EnvConfig{})
|
||||||
if err == nil {
|
if err == nil {
|
||||||
@@ -702,6 +852,14 @@ func TestSeriatimSubprocessHelper(t *testing.T) {
|
|||||||
case "normalize_report_missing":
|
case "normalize_report_missing":
|
||||||
writeSeriatimHelperFile(outputPath, `{"schema":"seriatim.intermediate.v1","segments":[]}`)
|
writeSeriatimHelperFile(outputPath, `{"schema":"seriatim.intermediate.v1","segments":[]}`)
|
||||||
os.Exit(0)
|
os.Exit(0)
|
||||||
|
case "render_success":
|
||||||
|
writeSeriatimHelperFile(outputPath, "# Rendered transcript\n\nHello.\n")
|
||||||
|
_, _ = os.Stdout.WriteString("seriatim helper render stdout\n")
|
||||||
|
_, _ = os.Stderr.WriteString("seriatim helper render stderr\n")
|
||||||
|
os.Exit(0)
|
||||||
|
case "render_empty_output":
|
||||||
|
writeSeriatimHelperFile(outputPath, "")
|
||||||
|
os.Exit(0)
|
||||||
default:
|
default:
|
||||||
_, _ = os.Stderr.WriteString(fmt.Sprintf("unknown helper mode %q\n", mode))
|
_, _ = os.Stderr.WriteString(fmt.Sprintf("unknown helper mode %q\n", mode))
|
||||||
os.Exit(2)
|
os.Exit(2)
|
||||||
@@ -732,7 +890,7 @@ func mergeReqForTest(t *testing.T, withReport bool) MergeRequest {
|
|||||||
req := MergeRequest{
|
req := MergeRequest{
|
||||||
GeneratedConfigPath: filepath.Join(dir, "seriatim.generated.yml"),
|
GeneratedConfigPath: filepath.Join(dir, "seriatim.generated.yml"),
|
||||||
InputTranscriptPaths: []string{in1, in2},
|
InputTranscriptPaths: []string{in1, in2},
|
||||||
OutputMergedTranscriptPath: filepath.Join(dir, "merged.json"),
|
OutputMergedTranscriptPath: filepath.Join(dir, "base.json"),
|
||||||
StdoutLogPath: filepath.Join(dir, "seriatim.stdout.log"),
|
StdoutLogPath: filepath.Join(dir, "seriatim.stdout.log"),
|
||||||
StderrLogPath: filepath.Join(dir, "seriatim.stderr.log"),
|
StderrLogPath: filepath.Join(dir, "seriatim.stderr.log"),
|
||||||
}
|
}
|
||||||
@@ -745,11 +903,11 @@ func mergeReqForTest(t *testing.T, withReport bool) MergeRequest {
|
|||||||
func trimReqForTest(t *testing.T) TrimRequest {
|
func trimReqForTest(t *testing.T) TrimRequest {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
dir := t.TempDir()
|
dir := t.TempDir()
|
||||||
input := filepath.Join(dir, "processed.json")
|
input := filepath.Join(dir, "polished.json")
|
||||||
writeSeriatimFile(t, input, `{"schema":"seriatim.intermediate.v1","segments":[]}`)
|
writeSeriatimFile(t, input, `{"schema":"seriatim.intermediate.v1","segments":[]}`)
|
||||||
return TrimRequest{
|
return TrimRequest{
|
||||||
InputTranscriptPath: input,
|
InputTranscriptPath: input,
|
||||||
OutputTrimmedPath: filepath.Join(dir, "trimmed.json"),
|
OutputTrimmedPath: filepath.Join(dir, "final.trimmed.json"),
|
||||||
KeepSelector: "5-12",
|
KeepSelector: "5-12",
|
||||||
GeneratedConfigPath: filepath.Join(dir, "seriatim.trim.generated.yml"),
|
GeneratedConfigPath: filepath.Join(dir, "seriatim.trim.generated.yml"),
|
||||||
StdoutLogPath: filepath.Join(dir, "seriatim.trim.stdout.log"),
|
StdoutLogPath: filepath.Join(dir, "seriatim.trim.stdout.log"),
|
||||||
@@ -760,12 +918,12 @@ func trimReqForTest(t *testing.T) TrimRequest {
|
|||||||
func normalizeReqForTest(t *testing.T, withReport bool) NormalizeRequest {
|
func normalizeReqForTest(t *testing.T, withReport bool) NormalizeRequest {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
dir := t.TempDir()
|
dir := t.TempDir()
|
||||||
input := filepath.Join(dir, "processed.json")
|
input := filepath.Join(dir, "polished.json")
|
||||||
writeSeriatimFile(t, input, `{"schema":"audita.processed.v1","segments":[]}`)
|
writeSeriatimFile(t, input, `{"schema":"audita.processed.v1","segments":[]}`)
|
||||||
|
|
||||||
req := NormalizeRequest{
|
req := NormalizeRequest{
|
||||||
InputTranscriptPath: input,
|
InputTranscriptPath: input,
|
||||||
OutputNormalizedPath: filepath.Join(dir, "normalized.json"),
|
OutputNormalizedPath: filepath.Join(dir, "final.json"),
|
||||||
OutputSchema: "seriatim-intermediate",
|
OutputSchema: "seriatim-intermediate",
|
||||||
GeneratedConfigPath: filepath.Join(dir, "seriatim.normalize.generated.yml"),
|
GeneratedConfigPath: filepath.Join(dir, "seriatim.normalize.generated.yml"),
|
||||||
StdoutLogPath: filepath.Join(dir, "seriatim.normalize.stdout.log"),
|
StdoutLogPath: filepath.Join(dir, "seriatim.normalize.stdout.log"),
|
||||||
@@ -777,6 +935,25 @@ func normalizeReqForTest(t *testing.T, withReport bool) NormalizeRequest {
|
|||||||
return req
|
return req
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func renderReqForTest(t *testing.T) RenderRequest {
|
||||||
|
t.Helper()
|
||||||
|
dir := t.TempDir()
|
||||||
|
input := filepath.Join(dir, "final.trimmed.json")
|
||||||
|
writeSeriatimFile(t, input, `{"schema":"seriatim.intermediate.v1","segments":[]}`)
|
||||||
|
return RenderRequest{
|
||||||
|
InputTranscriptPath: input,
|
||||||
|
OutputRenderedPath: filepath.Join(dir, "final.trimmed.md"),
|
||||||
|
Format: "markdown",
|
||||||
|
Title: "Session 42",
|
||||||
|
IncludeTimestamps: true,
|
||||||
|
IncludeSegmentIDs: true,
|
||||||
|
IncludeMetadata: false,
|
||||||
|
GeneratedConfigPath: filepath.Join(dir, "seriatim.render.generated.yml"),
|
||||||
|
StdoutLogPath: filepath.Join(dir, "seriatim.render.stdout.log"),
|
||||||
|
StderrLogPath: filepath.Join(dir, "seriatim.render.stderr.log"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func mustRunner(t *testing.T, binary string, report bool) *SubprocessRunner {
|
func mustRunner(t *testing.T, binary string, report bool) *SubprocessRunner {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
coalesce := 3.0
|
coalesce := 3.0
|
||||||
|
|||||||
@@ -1,31 +0,0 @@
|
|||||||
// Package storage declares archive/storage backend adapter boundaries.
|
|
||||||
package storage
|
|
||||||
|
|
||||||
import "context"
|
|
||||||
|
|
||||||
// TODO: implement remote storage/archive backends (S3/SFTP/etc.).
|
|
||||||
|
|
||||||
// Backend is the adapter boundary for archive/storage operations.
|
|
||||||
type Backend interface {
|
|
||||||
Archive(ctx context.Context, req ArchiveRequest) (ArchiveResult, error)
|
|
||||||
}
|
|
||||||
|
|
||||||
// ArchiveItem describes one item to archive.
|
|
||||||
type ArchiveItem struct {
|
|
||||||
Kind string
|
|
||||||
LocalPath string
|
|
||||||
RemoteKey string
|
|
||||||
}
|
|
||||||
|
|
||||||
// ArchiveRequest describes one archive operation.
|
|
||||||
type ArchiveRequest struct {
|
|
||||||
SessionID string
|
|
||||||
ManifestPath string
|
|
||||||
Items []ArchiveItem
|
|
||||||
}
|
|
||||||
|
|
||||||
// ArchiveResult describes archive operation output.
|
|
||||||
type ArchiveResult struct {
|
|
||||||
Archived []ArchiveItem
|
|
||||||
Metadata map[string]any
|
|
||||||
}
|
|
||||||
90
internal/adapters/storage/bounded_read.go
Normal file
90
internal/adapters/storage/bounded_read.go
Normal file
@@ -0,0 +1,90 @@
|
|||||||
|
package storage
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
|
"io"
|
||||||
|
"math"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
// ReadLimitError reports that a remote object exceeded its caller-owned read
|
||||||
|
// limit. The limit is enforced against both available object metadata and the
|
||||||
|
// bytes returned by the opened object body.
|
||||||
|
type ReadLimitError struct {
|
||||||
|
Key string
|
||||||
|
Limit int64
|
||||||
|
Observed int64
|
||||||
|
}
|
||||||
|
|
||||||
|
func (e *ReadLimitError) Error() string {
|
||||||
|
return fmt.Sprintf("object %q exceeds %d-byte read limit (observed at least %d bytes)", e.Key, e.Limit, e.Observed)
|
||||||
|
}
|
||||||
|
|
||||||
|
// ReadObjectBounded opens one object version and retains at most maxBytes of
|
||||||
|
// its content. Object metadata may reject an oversized body early, but a
|
||||||
|
// limit-plus-one read always enforces the boundary when transfer begins.
|
||||||
|
func ReadObjectBounded(ctx context.Context, store ObjectStore, key string, maxBytes int64) (info ObjectInfo, data []byte, err error) {
|
||||||
|
key = strings.TrimSpace(key)
|
||||||
|
if store == nil {
|
||||||
|
return ObjectInfo{}, nil, fmt.Errorf("read bounded object: store is required")
|
||||||
|
}
|
||||||
|
if key == "" {
|
||||||
|
return ObjectInfo{}, nil, fmt.Errorf("read bounded object: key is required")
|
||||||
|
}
|
||||||
|
if maxBytes <= 0 || maxBytes == math.MaxInt64 {
|
||||||
|
return ObjectInfo{}, nil, fmt.Errorf("read bounded object %q: limit must be between 1 and %d bytes", key, int64(math.MaxInt64-1))
|
||||||
|
}
|
||||||
|
if err := ctx.Err(); err != nil {
|
||||||
|
return ObjectInfo{}, nil, err
|
||||||
|
}
|
||||||
|
|
||||||
|
info, body, err := store.Read(ctx, key)
|
||||||
|
if err != nil {
|
||||||
|
return ObjectInfo{}, nil, err
|
||||||
|
}
|
||||||
|
if body == nil {
|
||||||
|
return ObjectInfo{}, nil, fmt.Errorf("read bounded object %q: store returned no body", key)
|
||||||
|
}
|
||||||
|
defer func() {
|
||||||
|
if closeErr := body.Close(); closeErr != nil {
|
||||||
|
data = nil
|
||||||
|
err = errors.Join(err, fmt.Errorf("close object %q: %w", key, closeErr))
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
|
||||||
|
if info.Size > maxBytes {
|
||||||
|
return info, nil, &ReadLimitError{Key: key, Limit: maxBytes, Observed: info.Size}
|
||||||
|
}
|
||||||
|
|
||||||
|
data, err = io.ReadAll(io.LimitReader(contextReader{ctx: ctx, reader: body}, maxBytes+1))
|
||||||
|
if err != nil {
|
||||||
|
return info, nil, err
|
||||||
|
}
|
||||||
|
if err := ctx.Err(); err != nil {
|
||||||
|
return info, nil, err
|
||||||
|
}
|
||||||
|
if int64(len(data)) > maxBytes {
|
||||||
|
return info, nil, &ReadLimitError{Key: key, Limit: maxBytes, Observed: int64(len(data))}
|
||||||
|
}
|
||||||
|
return info, data, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type contextReader struct {
|
||||||
|
ctx context.Context
|
||||||
|
reader io.Reader
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r contextReader) Read(p []byte) (int, error) {
|
||||||
|
if err := r.ctx.Err(); err != nil {
|
||||||
|
return 0, err
|
||||||
|
}
|
||||||
|
n, err := r.reader.Read(p)
|
||||||
|
if err == nil {
|
||||||
|
if contextErr := r.ctx.Err(); contextErr != nil {
|
||||||
|
return n, contextErr
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return n, err
|
||||||
|
}
|
||||||
135
internal/adapters/storage/bounded_read_test.go
Normal file
135
internal/adapters/storage/bounded_read_test.go
Normal file
@@ -0,0 +1,135 @@
|
|||||||
|
package storage
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"io"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestReadObjectBoundedAcceptsExactLimitWithAbsentSizeMetadata(t *testing.T) {
|
||||||
|
body := &trackingReadCloser{reader: bytes.NewReader([]byte("12345678")), chunkSize: 2}
|
||||||
|
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
|
||||||
|
return ObjectInfo{Key: "control.json", ETag: "generation"}, body, nil
|
||||||
|
}}
|
||||||
|
|
||||||
|
info, data, err := ReadObjectBounded(context.Background(), store, "control.json", 8)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ReadObjectBounded() error = %v", err)
|
||||||
|
}
|
||||||
|
if string(data) != "12345678" || info.ETag != "generation" {
|
||||||
|
t.Fatalf("ReadObjectBounded() = (%#v, %q), want opened object metadata and bytes", info, data)
|
||||||
|
}
|
||||||
|
if !body.closed {
|
||||||
|
t.Fatal("object body was not closed")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestReadObjectBoundedRejectsLimitPlusOneDespiteMissingOrInaccurateMetadata(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
metadataSize int64
|
||||||
|
}{
|
||||||
|
{name: "missing", metadataSize: 0},
|
||||||
|
{name: "inaccurate", metadataSize: 2},
|
||||||
|
}
|
||||||
|
for _, test := range tests {
|
||||||
|
t.Run(test.name, func(t *testing.T) {
|
||||||
|
body := &trackingReadCloser{reader: bytes.NewReader([]byte("123456789")), chunkSize: 1}
|
||||||
|
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
|
||||||
|
return ObjectInfo{Key: "control.json", Size: test.metadataSize}, body, nil
|
||||||
|
}}
|
||||||
|
|
||||||
|
_, data, err := ReadObjectBounded(context.Background(), store, "control.json", 8)
|
||||||
|
var limitErr *ReadLimitError
|
||||||
|
if !errors.As(err, &limitErr) {
|
||||||
|
t.Fatalf("ReadObjectBounded() error = %v, want ReadLimitError", err)
|
||||||
|
}
|
||||||
|
if data != nil || body.bytesRead != 9 || !body.closed {
|
||||||
|
t.Fatalf("data=%q bytes read=%d closed=%t, want nil, 9, true", data, body.bytesRead, body.closed)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestReadObjectBoundedRejectsOversizedMetadataBeforeTransfer(t *testing.T) {
|
||||||
|
body := &trackingReadCloser{reader: bytes.NewReader([]byte("small"))}
|
||||||
|
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
|
||||||
|
return ObjectInfo{Key: "control.json", Size: 9}, body, nil
|
||||||
|
}}
|
||||||
|
|
||||||
|
_, _, err := ReadObjectBounded(context.Background(), store, "control.json", 8)
|
||||||
|
var limitErr *ReadLimitError
|
||||||
|
if !errors.As(err, &limitErr) {
|
||||||
|
t.Fatalf("ReadObjectBounded() error = %v, want ReadLimitError", err)
|
||||||
|
}
|
||||||
|
if body.bytesRead != 0 || !body.closed {
|
||||||
|
t.Fatalf("bytes read=%d closed=%t, want zero-byte transfer and closed body", body.bytesRead, body.closed)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestReadObjectBoundedPropagatesCancellationAndClosesBody(t *testing.T) {
|
||||||
|
ctx, cancel := context.WithCancel(context.Background())
|
||||||
|
body := &trackingReadCloser{reader: bytes.NewReader([]byte("12345678")), chunkSize: 1, afterRead: cancel}
|
||||||
|
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
|
||||||
|
return ObjectInfo{Key: "control.json"}, body, nil
|
||||||
|
}}
|
||||||
|
|
||||||
|
_, data, err := ReadObjectBounded(ctx, store, "control.json", 8)
|
||||||
|
if !errors.Is(err, context.Canceled) {
|
||||||
|
t.Fatalf("ReadObjectBounded() error = %v, want context cancellation", err)
|
||||||
|
}
|
||||||
|
if data != nil || body.bytesRead != 1 || !body.closed {
|
||||||
|
t.Fatalf("data=%q bytes read=%d closed=%t, want nil, 1, true", data, body.bytesRead, body.closed)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestReadObjectBoundedReturnsCloseFailure(t *testing.T) {
|
||||||
|
closeErr := errors.New("close failed")
|
||||||
|
body := &trackingReadCloser{reader: bytes.NewReader([]byte("ok")), closeErr: closeErr}
|
||||||
|
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
|
||||||
|
return ObjectInfo{Key: "control.json", Size: 2}, body, nil
|
||||||
|
}}
|
||||||
|
|
||||||
|
_, data, err := ReadObjectBounded(context.Background(), store, "control.json", 8)
|
||||||
|
if !errors.Is(err, closeErr) || data != nil || !body.closed {
|
||||||
|
t.Fatalf("data=%q error=%v closed=%t, want close failure and no retained data", data, err, body.closed)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
type boundedReadStore struct {
|
||||||
|
ObjectStore
|
||||||
|
read func(context.Context, string) (ObjectInfo, io.ReadCloser, error)
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s *boundedReadStore) Read(ctx context.Context, key string) (ObjectInfo, io.ReadCloser, error) {
|
||||||
|
return s.read(ctx, key)
|
||||||
|
}
|
||||||
|
|
||||||
|
type trackingReadCloser struct {
|
||||||
|
reader io.Reader
|
||||||
|
chunkSize int
|
||||||
|
afterRead func()
|
||||||
|
closeErr error
|
||||||
|
bytesRead int
|
||||||
|
closed bool
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *trackingReadCloser) Read(p []byte) (int, error) {
|
||||||
|
if r.chunkSize > 0 && len(p) > r.chunkSize {
|
||||||
|
p = p[:r.chunkSize]
|
||||||
|
}
|
||||||
|
n, err := r.reader.Read(p)
|
||||||
|
r.bytesRead += n
|
||||||
|
if n > 0 && r.afterRead != nil {
|
||||||
|
r.afterRead()
|
||||||
|
r.afterRead = nil
|
||||||
|
}
|
||||||
|
return n, err
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *trackingReadCloser) Close() error {
|
||||||
|
r.closed = true
|
||||||
|
return r.closeErr
|
||||||
|
}
|
||||||
23
internal/adapters/storage/download_writer.go
Normal file
23
internal/adapters/storage/download_writer.go
Normal file
@@ -0,0 +1,23 @@
|
|||||||
|
package storage
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"fmt"
|
||||||
|
"io"
|
||||||
|
)
|
||||||
|
|
||||||
|
// WriterDownloader is implemented by storage backends that stream an object
|
||||||
|
// into a caller-owned file handle.
|
||||||
|
type WriterDownloader interface {
|
||||||
|
DownloadTo(ctx context.Context, key string, destination io.Writer) error
|
||||||
|
}
|
||||||
|
|
||||||
|
// DownloadTo streams one object into destination. Destination-confined callers
|
||||||
|
// require this capability rather than granting a backend a mutable pathname.
|
||||||
|
func DownloadTo(ctx context.Context, store ObjectStore, key string, destination io.Writer) error {
|
||||||
|
writer, ok := store.(WriterDownloader)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("object store does not support handle-confined downloads")
|
||||||
|
}
|
||||||
|
return writer.DownloadTo(ctx, key, destination)
|
||||||
|
}
|
||||||
@@ -14,16 +14,15 @@ func NewObjectStoreFromConfig(ctx context.Context, cfg *config.Config) (ObjectSt
|
|||||||
return nil, fmt.Errorf("pipeline config is required")
|
return nil, fmt.Errorf("pipeline config is required")
|
||||||
}
|
}
|
||||||
|
|
||||||
if strings.EqualFold(strings.TrimSpace(cfg.Pipeline.Storage.Backend), "s3") {
|
switch strings.ToLower(strings.TrimSpace(cfg.Pipeline.Storage.Backend)) {
|
||||||
|
case config.StorageBackendS3:
|
||||||
if cfg.Pipeline.Storage.S3 == nil {
|
if cfg.Pipeline.Storage.S3 == nil {
|
||||||
return nil, fmt.Errorf("pipeline.storage.s3 is required when pipeline.storage.backend is s3")
|
return nil, fmt.Errorf("pipeline.storage.s3 is required when pipeline.storage.backend is s3")
|
||||||
}
|
}
|
||||||
return NewS3BackendFromConfig(ctx, *cfg.Pipeline.Storage.S3)
|
return NewS3BackendFromConfig(ctx, *cfg.Pipeline.Storage.S3)
|
||||||
|
case "", config.StorageBackendLocal:
|
||||||
|
return nil, fmt.Errorf("no remote object store backend is configured")
|
||||||
|
default:
|
||||||
|
return nil, fmt.Errorf("unsupported pipeline.storage.backend %q", cfg.Pipeline.Storage.Backend)
|
||||||
}
|
}
|
||||||
|
|
||||||
if cfg.Pipeline.Storage.S3 != nil && strings.TrimSpace(cfg.Pipeline.Storage.S3.Bucket) != "" {
|
|
||||||
return NewS3BackendFromConfig(ctx, *cfg.Pipeline.Storage.S3)
|
|
||||||
}
|
|
||||||
|
|
||||||
return nil, fmt.Errorf("no remote object store backend is configured")
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -54,3 +54,35 @@ func TestNewObjectStoreFromConfigNoRemoteBackendConfigured(t *testing.T) {
|
|||||||
t.Fatalf("NewObjectStoreFromConfig() error = %v, want no-backend error", err)
|
t.Fatalf("NewObjectStoreFromConfig() error = %v, want no-backend error", err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestNewObjectStoreFromConfigDoesNotInferS3FromProviderFields(t *testing.T) {
|
||||||
|
called := false
|
||||||
|
original := newS3Client
|
||||||
|
t.Cleanup(func() { newS3Client = original })
|
||||||
|
newS3Client = func(_ context.Context, _ s3ClientOptions) (s3API, error) {
|
||||||
|
called = true
|
||||||
|
return &fakeS3API{}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
_, err := NewObjectStoreFromConfig(context.Background(), &config.Config{
|
||||||
|
Pipeline: &config.PipelineConfig{Storage: config.StorageConfig{
|
||||||
|
Backend: config.StorageBackendLocal,
|
||||||
|
S3: &config.StorageS3Config{Bucket: "my-archive"},
|
||||||
|
}},
|
||||||
|
})
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "no remote object store backend is configured") {
|
||||||
|
t.Fatalf("NewObjectStoreFromConfig() error = %v, want no-backend error", err)
|
||||||
|
}
|
||||||
|
if called {
|
||||||
|
t.Fatal("NewObjectStoreFromConfig() constructed S3 from incidental provider fields")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestNewObjectStoreFromConfigRejectsUnknownBackend(t *testing.T) {
|
||||||
|
_, err := NewObjectStoreFromConfig(context.Background(), &config.Config{
|
||||||
|
Pipeline: &config.PipelineConfig{Storage: config.StorageConfig{Backend: "s33"}},
|
||||||
|
})
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "unsupported pipeline.storage.backend") {
|
||||||
|
t.Fatalf("NewObjectStoreFromConfig() error = %v, want unsupported-backend error", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user